@tangle-network/agent-eval 0.144.5 → 0.144.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -0
- package/dist/{benchmark-Fmo42QVE.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +471 -89
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CA_NFOmy.js → benchmark-command-BCafwNrf.js} +664 -605
- package/dist/benchmark-command-BCafwNrf.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CWbsj0t4.js → benchmarks-CDSolHq7.js} +4 -4
- package/dist/{benchmarks-CWbsj0t4.js.map → benchmarks-CDSolHq7.js.map} +1 -1
- package/dist/campaign/index.d.ts +8 -6
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-ClpnD7Ug.js → campaign-Tdy3h62h.js} +17 -301
- package/dist/campaign-Tdy3h62h.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-DAb7MWtL.d.ts → client-DjXROWpx.d.ts} +4 -4
- package/dist/{client-DAb7MWtL.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-VvpHRu78.d.ts → completion-verifier-foUCLif_.d.ts} +6 -6
- package/dist/{completion-verifier-VvpHRu78.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +11 -10
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
- package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
- package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
- package/dist/{default-registry-BwbZ9N9v.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-RLNNoeEP.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-RLNNoeEP.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-ByZiqpVk.d.ts → engine-nB64f48I.d.ts} +21 -34
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
- package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
- package/dist/errors-D-LKuDhb.js.map +1 -1
- package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-BQ7W90C4.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-BQ7W90C4.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts → feedback-trajectory-Rh280oXo.d.ts} +4 -4
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-CsuAo2-J.d.ts → index-C5HOo4ZF2.d.ts} +5 -5
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
- package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
- package/dist/{index-CNOCxBLh.d.ts → index-CvXXlyz7.d.ts} +3 -3
- package/dist/{index-CNOCxBLh.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-DGIzNtRv.d.ts → index-CwDrUMe0.d.ts} +3 -3
- package/dist/{index-DGIzNtRv.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-CpxZSlB7.d.ts → index-Sh2I0DRc.d.ts} +15 -649
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +252 -451
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +271 -751
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-C6h6F_4L.d.ts} +4 -4
- package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-DRXobPEs.d.ts → integrity-BuqEKu-x.d.ts} +3 -3
- package/dist/{integrity-DRXobPEs.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/{llm-client-D3EoChAU.js → llm-client-DzvMUsS_.js} +310 -10
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +3 -3
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-BZRWdq_t.d.ts → release-report-CI8uisI1.d.ts} +4 -4
- package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-B7S7Pdbw.d.ts → replay-DFf-teiC.d.ts} +8 -7
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +2 -2
- package/dist/{researcher-BSiCoM1s.d.ts → researcher-BoaxeCzP.d.ts} +6 -6
- package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-Cf1PtEOz.d.ts} +34 -4
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +19 -9
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
- package/dist/{run-evidence-j5Ynww6L.d.ts → run-evidence-BDFFai9R.d.ts} +3 -3
- package/dist/{run-evidence-j5Ynww6L.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
- package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DKRtp2sY.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DKRtp2sY.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-Df00sdwz.js → server-iu0ede49.js} +2 -2
- package/dist/{server-Df00sdwz.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-DUvvudWR.d.ts → skill-usage-CJlWEUFt.d.ts} +11 -11
- package/dist/{skill-usage-DUvvudWR.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-CGz9ywhM.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +11 -297
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
- package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-B4u_CiFd.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-DuUS_i7W.d.ts} +4 -115
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-BjMFz88h.d.ts → types-D216SgwM.d.ts} +228 -9
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-CZt1PBIk.d.ts → types-DF_Udrp-.d.ts} +54 -5
- package/dist/{types-CZt1PBIk.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-Dcoaqcsc.d.ts → types-DYuNHo9R.d.ts} +5 -5
- package/dist/{types-Dcoaqcsc.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/building-doctrine.md +15 -0
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +21 -4
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-Fmo42QVE.d.ts.map +0 -1
- package/dist/benchmark-command-CA_NFOmy.js.map +0 -1
- package/dist/campaign-ClpnD7Ug.js.map +0 -1
- package/dist/default-registry-BwbZ9N9v.d.ts.map +0 -1
- package/dist/index-CpxZSlB7.d.ts.map +0 -1
- package/dist/index-CsuAo2-J.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-D3EoChAU.js.map +0 -1
- package/dist/replay-B7S7Pdbw.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-DFgkEY4p.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-C4FX42dy.js.map +0 -1
- package/dist/skillopt-optimization-method-CGz9ywhM.d.ts.map +0 -1
- package/dist/summary-report-BOM6dfP7.d.ts.map +0 -1
- package/dist/tool-groups-ByZiqpVk.d.ts.map +0 -1
- package/dist/types-BjMFz88h.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
|
@@ -0,0 +1,2103 @@
|
|
|
1
|
+
import { t as TraceEmitter } from "../emitter-CPBAhxum.js";
|
|
2
|
+
import { n as InMemoryTraceStore } from "../store-DNe_Uv1Q.js";
|
|
3
|
+
import { n as runCounterfactual } from "../counterfactual-CWPTrMH7.js";
|
|
4
|
+
import { a as parseObservationOutput, i as isSubmitAction, n as SUBMIT_ACTION_SIGNATURE, o as parseRecordedReturncode, r as deriveFailureSignature, t as wrapActionForExec } from "../exec-BLtYZdWo.js";
|
|
5
|
+
import { appendFileSync, existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, writeFileSync } from "node:fs";
|
|
6
|
+
import { join } from "node:path";
|
|
7
|
+
import { createHash } from "node:crypto";
|
|
8
|
+
import { execFile } from "node:child_process";
|
|
9
|
+
import { tmpdir } from "node:os";
|
|
10
|
+
import { promisify } from "node:util";
|
|
11
|
+
//#region src/trajectory-replay/corpus.ts
|
|
12
|
+
/**
|
|
13
|
+
* Corpus enumeration for batch replay verification.
|
|
14
|
+
*
|
|
15
|
+
* A labeled trajectory corpus is a gold label file (array of trajectory
|
|
16
|
+
* entries with `incorrect_stages[].incorrect_step_ids`) plus a prepared
|
|
17
|
+
* directory:
|
|
18
|
+
* <prepared>/normalized/<traj_id>/steps.json normalized steps
|
|
19
|
+
* <prepared>/normalized/<traj_id>/task.md optional task statement
|
|
20
|
+
* <prepared>/extracted/<traj_id>/swe_raw/** raw mini-SWE trajectory
|
|
21
|
+
*
|
|
22
|
+
* Only trajectories that record their environment are replayable: the raw
|
|
23
|
+
* trajectory must carry `info.docker_config.base_image`. Every excluded case
|
|
24
|
+
* is returned with a machine-readable reason — the batch report surfaces the
|
|
25
|
+
* full exclusion table, never a silently shrunk denominator.
|
|
26
|
+
*/
|
|
27
|
+
/** Parses `name=<labelsPath>::<preparedDir>` (paths may contain `=`, not `::`). */
|
|
28
|
+
function parseCorpusFlag(value) {
|
|
29
|
+
const eq = value.indexOf("=");
|
|
30
|
+
if (eq <= 0) throw new Error(`--corpus must be name=<labels>::<prepared>, got: ${value}`);
|
|
31
|
+
const name = value.slice(0, eq);
|
|
32
|
+
const rest = value.slice(eq + 1);
|
|
33
|
+
const sep = rest.indexOf("::");
|
|
34
|
+
if (sep <= 0 || sep === rest.length - 2) throw new Error(`--corpus must be name=<labels>::<prepared>, got: ${value}`);
|
|
35
|
+
return {
|
|
36
|
+
name,
|
|
37
|
+
labelsPath: rest.slice(0, sep),
|
|
38
|
+
preparedDir: rest.slice(sep + 2)
|
|
39
|
+
};
|
|
40
|
+
}
|
|
41
|
+
function readLabelEntries(labelsPath) {
|
|
42
|
+
const parsed = JSON.parse(readFileSync(labelsPath, "utf8"));
|
|
43
|
+
if (!Array.isArray(parsed)) throw new Error(`${labelsPath} is not a JSON array of label entries`);
|
|
44
|
+
for (const entry of parsed) if (typeof entry.traj_id !== "string") throw new Error(`${labelsPath} has an entry without a string traj_id`);
|
|
45
|
+
return parsed;
|
|
46
|
+
}
|
|
47
|
+
function goldIncorrectSteps(entry) {
|
|
48
|
+
const ids = /* @__PURE__ */ new Set();
|
|
49
|
+
for (const stage of entry.incorrect_stages ?? []) for (const id of stage.incorrect_step_ids ?? []) ids.add(id);
|
|
50
|
+
return [...ids].sort((a, b) => a - b);
|
|
51
|
+
}
|
|
52
|
+
function findRawTrajFiles(sweRawDir) {
|
|
53
|
+
if (!existsSync(sweRawDir)) return [];
|
|
54
|
+
return readdirSync(sweRawDir, {
|
|
55
|
+
recursive: true,
|
|
56
|
+
encoding: "utf8"
|
|
57
|
+
}).filter((relative) => relative.endsWith(".traj.json")).map((relative) => join(sweRawDir, relative)).sort();
|
|
58
|
+
}
|
|
59
|
+
/** First non-empty output line of the trajectory's first bare `pwd` step. */
|
|
60
|
+
function cwdFromPwdObservation(steps) {
|
|
61
|
+
for (const step of steps) {
|
|
62
|
+
if (step.action.trim() !== "pwd") continue;
|
|
63
|
+
const line = parseObservationOutput(step.observation).split("\n").map((l) => l.trim()).find((l) => l.startsWith("/"));
|
|
64
|
+
if (line) return line;
|
|
65
|
+
}
|
|
66
|
+
return null;
|
|
67
|
+
}
|
|
68
|
+
function taskStatementOf(preparedDir, trajId, raw) {
|
|
69
|
+
const taskPath = join(preparedDir, "normalized", trajId, "task.md");
|
|
70
|
+
if (existsSync(taskPath)) {
|
|
71
|
+
const text = readFileSync(taskPath, "utf8").trim();
|
|
72
|
+
if (text.length > 0) return text;
|
|
73
|
+
}
|
|
74
|
+
const userMessage = raw.messages?.find((m) => m.role === "user");
|
|
75
|
+
if (typeof userMessage?.content === "string" && userMessage.content.trim().length > 0) return userMessage.content.trim();
|
|
76
|
+
return null;
|
|
77
|
+
}
|
|
78
|
+
/**
|
|
79
|
+
* Resolves the replay resources for one trajectory, independent of gold
|
|
80
|
+
* labels — the finding wire uses this with a finding-supplied step instead.
|
|
81
|
+
*/
|
|
82
|
+
function resolveCaseResources(corpus, trajId) {
|
|
83
|
+
const sweRawDir = join(corpus.preparedDir, "extracted", trajId, "swe_raw");
|
|
84
|
+
const rawFiles = findRawTrajFiles(sweRawDir);
|
|
85
|
+
if (rawFiles.length === 0) return {
|
|
86
|
+
resolved: false,
|
|
87
|
+
reason: "no-swe-raw-trajectory",
|
|
88
|
+
detail: sweRawDir
|
|
89
|
+
};
|
|
90
|
+
if (rawFiles.length > 1) return {
|
|
91
|
+
resolved: false,
|
|
92
|
+
reason: "ambiguous-swe-raw-trajectory",
|
|
93
|
+
detail: rawFiles.join(", ")
|
|
94
|
+
};
|
|
95
|
+
let raw;
|
|
96
|
+
try {
|
|
97
|
+
raw = JSON.parse(readFileSync(rawFiles[0], "utf8"));
|
|
98
|
+
} catch (err) {
|
|
99
|
+
return {
|
|
100
|
+
resolved: false,
|
|
101
|
+
reason: "unreadable-raw-trajectory",
|
|
102
|
+
detail: `${rawFiles[0]}: ${err instanceof Error ? err.message : String(err)}`
|
|
103
|
+
};
|
|
104
|
+
}
|
|
105
|
+
const dockerConfig = raw.info?.docker_config;
|
|
106
|
+
const image = dockerConfig?.base_image;
|
|
107
|
+
if (typeof image !== "string" || image.length === 0) return {
|
|
108
|
+
resolved: false,
|
|
109
|
+
reason: "no-docker-image",
|
|
110
|
+
detail: rawFiles[0]
|
|
111
|
+
};
|
|
112
|
+
const stepsPath = join(corpus.preparedDir, "normalized", trajId, "steps.json");
|
|
113
|
+
if (!existsSync(stepsPath)) return {
|
|
114
|
+
resolved: false,
|
|
115
|
+
reason: "missing-steps-json",
|
|
116
|
+
detail: stepsPath
|
|
117
|
+
};
|
|
118
|
+
const steps = JSON.parse(readFileSync(stepsPath, "utf8"));
|
|
119
|
+
const runConfigCwd = raw.info?.config?.environment?.cwd;
|
|
120
|
+
const dockerCwd = dockerConfig?.cwd;
|
|
121
|
+
let cwd;
|
|
122
|
+
let cwdSource;
|
|
123
|
+
if (typeof runConfigCwd === "string" && runConfigCwd.length > 0) {
|
|
124
|
+
cwd = runConfigCwd;
|
|
125
|
+
cwdSource = "run-config";
|
|
126
|
+
} else if (typeof dockerCwd === "string" && dockerCwd.length > 0) {
|
|
127
|
+
cwd = dockerCwd;
|
|
128
|
+
cwdSource = "docker-config";
|
|
129
|
+
} else {
|
|
130
|
+
const observed = cwdFromPwdObservation(steps);
|
|
131
|
+
if (!observed) return {
|
|
132
|
+
resolved: false,
|
|
133
|
+
reason: "cwd-underivable"
|
|
134
|
+
};
|
|
135
|
+
cwd = observed;
|
|
136
|
+
cwdSource = "pwd-observation";
|
|
137
|
+
}
|
|
138
|
+
const recordedTimeout = raw.info?.config?.environment?.timeout;
|
|
139
|
+
return {
|
|
140
|
+
resolved: true,
|
|
141
|
+
resources: {
|
|
142
|
+
corpus: corpus.name,
|
|
143
|
+
trajId,
|
|
144
|
+
stepsPath,
|
|
145
|
+
steps,
|
|
146
|
+
taskStatement: taskStatementOf(corpus.preparedDir, trajId, raw),
|
|
147
|
+
image,
|
|
148
|
+
cwd,
|
|
149
|
+
cwdSource,
|
|
150
|
+
recordedStepTimeoutMs: typeof recordedTimeout === "number" && recordedTimeout > 0 ? recordedTimeout * 1e3 : null
|
|
151
|
+
}
|
|
152
|
+
};
|
|
153
|
+
}
|
|
154
|
+
/**
|
|
155
|
+
* Replayable = a raw trajectory with a recorded image AND at least one gold
|
|
156
|
+
* incorrect step that is a real mid-trajectory action. Gold steps whose action
|
|
157
|
+
* is the submit command are skipped when choosing k (counted per case); a case
|
|
158
|
+
* whose golds are ALL submit steps is excluded as gold-only-submit-step.
|
|
159
|
+
* Exclusion reasons are reported in resolution order:
|
|
160
|
+
* raw trajectory → image → gold labels → steps.json → k range → cwd.
|
|
161
|
+
*/
|
|
162
|
+
function enumerateReplayableCases(corpora) {
|
|
163
|
+
const replayable = [];
|
|
164
|
+
const excluded = [];
|
|
165
|
+
let labelEntryCount = 0;
|
|
166
|
+
for (const corpus of corpora) for (const entry of readLabelEntries(corpus.labelsPath)) {
|
|
167
|
+
labelEntryCount += 1;
|
|
168
|
+
const trajId = entry.traj_id;
|
|
169
|
+
const resolution = resolveCaseResources(corpus, trajId);
|
|
170
|
+
if (!resolution.resolved) {
|
|
171
|
+
excluded.push({
|
|
172
|
+
corpus: corpus.name,
|
|
173
|
+
trajId,
|
|
174
|
+
reason: resolution.reason,
|
|
175
|
+
detail: resolution.detail
|
|
176
|
+
});
|
|
177
|
+
continue;
|
|
178
|
+
}
|
|
179
|
+
const gold = goldIncorrectSteps(entry);
|
|
180
|
+
if (gold.length === 0) {
|
|
181
|
+
excluded.push({
|
|
182
|
+
corpus: corpus.name,
|
|
183
|
+
trajId,
|
|
184
|
+
reason: "no-gold-incorrect-step"
|
|
185
|
+
});
|
|
186
|
+
continue;
|
|
187
|
+
}
|
|
188
|
+
const resources = resolution.resources;
|
|
189
|
+
let submitGoldsSkipped = 0;
|
|
190
|
+
let target = null;
|
|
191
|
+
let missingGoldId = null;
|
|
192
|
+
for (const goldId of gold) {
|
|
193
|
+
const step = resources.steps.find((s) => s.step_id === goldId);
|
|
194
|
+
if (!step) {
|
|
195
|
+
missingGoldId = goldId;
|
|
196
|
+
break;
|
|
197
|
+
}
|
|
198
|
+
if (isSubmitAction(step.action)) {
|
|
199
|
+
submitGoldsSkipped += 1;
|
|
200
|
+
continue;
|
|
201
|
+
}
|
|
202
|
+
target = step;
|
|
203
|
+
break;
|
|
204
|
+
}
|
|
205
|
+
if (missingGoldId !== null) {
|
|
206
|
+
excluded.push({
|
|
207
|
+
corpus: corpus.name,
|
|
208
|
+
trajId,
|
|
209
|
+
reason: "gold-step-outside-steps",
|
|
210
|
+
detail: `k=${missingGoldId}, steps=${resources.steps.length}`
|
|
211
|
+
});
|
|
212
|
+
continue;
|
|
213
|
+
}
|
|
214
|
+
if (!target) {
|
|
215
|
+
excluded.push({
|
|
216
|
+
corpus: corpus.name,
|
|
217
|
+
trajId,
|
|
218
|
+
reason: "gold-only-submit-step",
|
|
219
|
+
detail: `${submitGoldsSkipped} gold step(s), all submit commands`
|
|
220
|
+
});
|
|
221
|
+
continue;
|
|
222
|
+
}
|
|
223
|
+
replayable.push({
|
|
224
|
+
...resources,
|
|
225
|
+
goldIncorrectSteps: gold,
|
|
226
|
+
k: target.step_id,
|
|
227
|
+
submitGoldsSkipped,
|
|
228
|
+
recordedReturncodeAtK: parseRecordedReturncode(target.observation)
|
|
229
|
+
});
|
|
230
|
+
}
|
|
231
|
+
return {
|
|
232
|
+
replayable,
|
|
233
|
+
excluded,
|
|
234
|
+
labelEntryCount
|
|
235
|
+
};
|
|
236
|
+
}
|
|
237
|
+
//#endregion
|
|
238
|
+
//#region src/trajectory-replay/fix.ts
|
|
239
|
+
/**
|
|
240
|
+
* Counterfactual patch synthesis: turn a recorded incorrect step into a
|
|
241
|
+
* corrected shell command for replay arm B.
|
|
242
|
+
*
|
|
243
|
+
* The caller is a typed outcome boundary ({ succeeded, value, error }) so a
|
|
244
|
+
* provider failure is a per-case report row, never a thrown batch abort and
|
|
245
|
+
* never a silent empty fix. Concrete chat transports live with the consumer;
|
|
246
|
+
* this module only builds prompts and reads replies.
|
|
247
|
+
*/
|
|
248
|
+
/** Head+tail excerpt with an elision marker; identity below the limit. */
|
|
249
|
+
function clipText(text, limit) {
|
|
250
|
+
if (text.length <= limit) return text;
|
|
251
|
+
const half = Math.floor(limit / 2);
|
|
252
|
+
return `${text.slice(0, half)}\n… [${text.length - limit} chars elided] …\n${text.slice(-half)}`;
|
|
253
|
+
}
|
|
254
|
+
function renderStep(step, marker) {
|
|
255
|
+
const rc = parseRecordedReturncode(step.observation);
|
|
256
|
+
const output = clipText(parseObservationOutput(step.observation).trim(), 1600);
|
|
257
|
+
return [
|
|
258
|
+
`### step ${step.step_id}${marker}`,
|
|
259
|
+
"```sh",
|
|
260
|
+
clipText(step.action, 2e3),
|
|
261
|
+
"```",
|
|
262
|
+
`returncode: ${rc ?? "none recorded"}`,
|
|
263
|
+
output.length > 0 ? `output:\n\`\`\`\n${output}\n\`\`\`` : "output: (empty)"
|
|
264
|
+
].join("\n");
|
|
265
|
+
}
|
|
266
|
+
/** Shared user-prompt sections: task statement, ±radius context, failing step. */
|
|
267
|
+
function promptBody(input) {
|
|
268
|
+
const radius = input.contextRadius ?? 3;
|
|
269
|
+
const target = input.steps.find((s) => s.step_id === input.k);
|
|
270
|
+
if (!target) throw new Error(`trajectory-replay: no step with step_id ${input.k}`);
|
|
271
|
+
const context = input.steps.filter((s) => s.step_id !== input.k && Math.abs(s.step_id - input.k) <= radius);
|
|
272
|
+
return [
|
|
273
|
+
"## Task the agent was solving",
|
|
274
|
+
input.taskStatement ? clipText(input.taskStatement, 4e3) : "(no task statement recorded)",
|
|
275
|
+
"",
|
|
276
|
+
"## Surrounding steps",
|
|
277
|
+
...context.map((s) => renderStep(s, "")),
|
|
278
|
+
"",
|
|
279
|
+
"## Failing step to correct",
|
|
280
|
+
renderStep(target, " (INCORRECT — correct this one)")
|
|
281
|
+
];
|
|
282
|
+
}
|
|
283
|
+
function buildFixPrompt(input) {
|
|
284
|
+
return {
|
|
285
|
+
system: [
|
|
286
|
+
"You repair one failed shell command from a recorded coding-agent trajectory.",
|
|
287
|
+
"The trajectory replays inside the original docker image; every command runs as a fresh /bin/sh subshell from a fixed working directory.",
|
|
288
|
+
"You are given the failing step, its recorded output, surrounding steps, and the task statement.",
|
|
289
|
+
"Reply with exactly ONE corrected shell command (compound commands with && or pipes are fine) inside a single ```sh fenced block.",
|
|
290
|
+
"The corrected command must accomplish the failing step's intent and exit 0. No prose outside the fenced block."
|
|
291
|
+
].join("\n"),
|
|
292
|
+
user: [
|
|
293
|
+
...promptBody(input),
|
|
294
|
+
"",
|
|
295
|
+
"Output the single corrected replacement for the failing step now."
|
|
296
|
+
].join("\n")
|
|
297
|
+
};
|
|
298
|
+
}
|
|
299
|
+
function renderFailedAttempt(prior) {
|
|
300
|
+
if (prior.command === null) return [`### attempt ${prior.attempt}`, `model call failed before producing a command: ${prior.llmError ?? "unknown error"}`].join("\n");
|
|
301
|
+
const stdout = (prior.stdoutTail ?? "").trim();
|
|
302
|
+
const stderr = (prior.stderrTail ?? "").trim();
|
|
303
|
+
return [
|
|
304
|
+
`### attempt ${prior.attempt}`,
|
|
305
|
+
"```sh",
|
|
306
|
+
clipText(prior.command, 2e3),
|
|
307
|
+
"```",
|
|
308
|
+
`exit code: ${prior.exitCode ?? "not executed"}`,
|
|
309
|
+
stdout.length > 0 ? `stdout:\n\`\`\`\n${stdout}\n\`\`\`` : "stdout: (empty)",
|
|
310
|
+
stderr.length > 0 ? `stderr:\n\`\`\`\n${stderr}\n\`\`\`` : "stderr: (empty)"
|
|
311
|
+
].join("\n");
|
|
312
|
+
}
|
|
313
|
+
/**
|
|
314
|
+
* Retry prompt for fix-loop attempts ≥2: the original context plus every prior
|
|
315
|
+
* attempt with its REAL executed output, and permission to answer with a short
|
|
316
|
+
* script (the block still executes as one /bin/sh unit).
|
|
317
|
+
*/
|
|
318
|
+
function buildRetryFixPrompt(input, priorAttempts, maxScriptCommands = 5) {
|
|
319
|
+
if (priorAttempts.length === 0) throw new Error("trajectory-replay: buildRetryFixPrompt requires at least one prior attempt");
|
|
320
|
+
return {
|
|
321
|
+
system: [
|
|
322
|
+
"You repair one failed shell command from a recorded coding-agent trajectory.",
|
|
323
|
+
"The trajectory replays inside the original docker image; every command runs as a fresh /bin/sh subshell from a fixed working directory.",
|
|
324
|
+
"Earlier corrected commands were executed for real and failed; their actual output is included below.",
|
|
325
|
+
`Reply with a corrected fix inside a single \`\`\`sh fenced block: either one command, or a short script of at most ${maxScriptCommands} commands (one per line).`,
|
|
326
|
+
"The whole block executes as ONE /bin/sh unit from the fixed working directory and must exit 0.",
|
|
327
|
+
"Do not repeat a command that already failed. Keep reasoning brief. No prose outside the fenced block."
|
|
328
|
+
].join("\n"),
|
|
329
|
+
user: [
|
|
330
|
+
...promptBody(input),
|
|
331
|
+
"",
|
|
332
|
+
"## Previous fix attempts (executed for real — all failed)",
|
|
333
|
+
...priorAttempts.map((prior) => renderFailedAttempt(prior)),
|
|
334
|
+
"",
|
|
335
|
+
`Output a corrected fix now — one \`\`\`sh block, at most ${maxScriptCommands} commands.`
|
|
336
|
+
].join("\n")
|
|
337
|
+
};
|
|
338
|
+
}
|
|
339
|
+
/** Non-empty, non-comment lines of a fix script — the loop's script-size cap. */
|
|
340
|
+
function countScriptCommands(script) {
|
|
341
|
+
return script.split("\n").filter((line) => line.trim().length > 0 && !line.trim().startsWith("#")).length;
|
|
342
|
+
}
|
|
343
|
+
/** Last fenced code block, else the whole trimmed content; null when empty. */
|
|
344
|
+
function extractFixCommand(content) {
|
|
345
|
+
const blocks = [...content.matchAll(/```(?:sh|bash|shell)?\n([\s\S]*?)```/g)];
|
|
346
|
+
const command = (blocks.length > 0 ? blocks[blocks.length - 1][1] : content).trim();
|
|
347
|
+
if (command.length === 0 || command.includes("```")) return null;
|
|
348
|
+
return command;
|
|
349
|
+
}
|
|
350
|
+
async function generateFixCommand(caller, input) {
|
|
351
|
+
const { system, user } = buildFixPrompt(input);
|
|
352
|
+
const outcome = await caller.complete(system, user);
|
|
353
|
+
if (!outcome.succeeded) return outcome;
|
|
354
|
+
const command = extractFixCommand(outcome.value.content);
|
|
355
|
+
if (command === null) return {
|
|
356
|
+
succeeded: false,
|
|
357
|
+
error: `completion carried no usable command: ${clipText(outcome.value.content, 300)}`
|
|
358
|
+
};
|
|
359
|
+
return {
|
|
360
|
+
succeeded: true,
|
|
361
|
+
value: {
|
|
362
|
+
command,
|
|
363
|
+
usage: outcome.value.usage
|
|
364
|
+
}
|
|
365
|
+
};
|
|
366
|
+
}
|
|
367
|
+
//#endregion
|
|
368
|
+
//#region src/trajectory-replay/fix-loop.ts
|
|
369
|
+
/**
|
|
370
|
+
* Iterative counterfactual fix loop.
|
|
371
|
+
*
|
|
372
|
+
* Attempt 1 is exactly the one-shot generator (same prompt, same caller), so
|
|
373
|
+
* flip@1 stays comparable to one-shot fix mode. When an attempt does not flip
|
|
374
|
+
* the failure — nonzero exit, signature still present, or the model call
|
|
375
|
+
* itself failed — the next attempt's prompt carries every prior command and
|
|
376
|
+
* its REAL executed stdout/stderr, up to a fixed attempt budget.
|
|
377
|
+
*
|
|
378
|
+
* Isolation invariant: every attempt executes through the injected executor,
|
|
379
|
+
* which must provide a FRESH sandbox with the same replayed prefix. A used
|
|
380
|
+
* sandbox is never mutated mid-arm; a flip therefore always proves the
|
|
381
|
+
* corrected step against the recorded prefix state, not against debris from
|
|
382
|
+
* an earlier attempt.
|
|
383
|
+
*/
|
|
384
|
+
function toFailedAttempt(record) {
|
|
385
|
+
return {
|
|
386
|
+
attempt: record.attempt,
|
|
387
|
+
command: record.command,
|
|
388
|
+
exitCode: record.exitCode,
|
|
389
|
+
stdoutTail: record.stdoutTail,
|
|
390
|
+
stderrTail: record.stderrTail,
|
|
391
|
+
llmError: record.llmError ?? record.armBError
|
|
392
|
+
};
|
|
393
|
+
}
|
|
394
|
+
async function runFixLoop(caller, input, executor, options) {
|
|
395
|
+
if (!Number.isInteger(options.maxAttempts) || options.maxAttempts < 1) throw new Error(`trajectory-replay: maxAttempts must be a positive integer, got ${options.maxAttempts}`);
|
|
396
|
+
const scriptCap = options.maxScriptCommands ?? 5;
|
|
397
|
+
const tailChars = options.outputTailChars ?? 1600;
|
|
398
|
+
const attempts = [];
|
|
399
|
+
let llmCalls = 0;
|
|
400
|
+
let llmFailures = 0;
|
|
401
|
+
let promptTokens = 0;
|
|
402
|
+
let completionTokens = 0;
|
|
403
|
+
let callsWithoutUsage = 0;
|
|
404
|
+
const unexecuted = (attempt, llmError, usage, command, started) => ({
|
|
405
|
+
attempt,
|
|
406
|
+
command,
|
|
407
|
+
llmError,
|
|
408
|
+
usage,
|
|
409
|
+
executed: false,
|
|
410
|
+
exitCode: null,
|
|
411
|
+
prefixExecuted: null,
|
|
412
|
+
prefixDivergences: null,
|
|
413
|
+
prefixDivergencePct: null,
|
|
414
|
+
failureVanished: null,
|
|
415
|
+
stdoutTail: null,
|
|
416
|
+
stderrTail: null,
|
|
417
|
+
armBError: null,
|
|
418
|
+
wallMs: Date.now() - started
|
|
419
|
+
});
|
|
420
|
+
for (let attempt = 1; attempt <= options.maxAttempts; attempt++) {
|
|
421
|
+
const started = Date.now();
|
|
422
|
+
const prompt = attempt === 1 ? buildFixPrompt(input) : buildRetryFixPrompt(input, attempts.map(toFailedAttempt), scriptCap);
|
|
423
|
+
llmCalls += 1;
|
|
424
|
+
const outcome = await caller.complete(prompt.system, prompt.user);
|
|
425
|
+
if (!outcome.succeeded) {
|
|
426
|
+
llmFailures += 1;
|
|
427
|
+
attempts.push(unexecuted(attempt, outcome.error, null, null, started));
|
|
428
|
+
options.onProgress?.(`fix-loop attempt ${attempt}: LLM failed — ${outcome.error.slice(0, 160)}`);
|
|
429
|
+
continue;
|
|
430
|
+
}
|
|
431
|
+
const usage = outcome.value.usage;
|
|
432
|
+
if (usage === null || usage === void 0) callsWithoutUsage += 1;
|
|
433
|
+
promptTokens += usage?.promptTokens ?? 0;
|
|
434
|
+
completionTokens += usage?.completionTokens ?? 0;
|
|
435
|
+
const command = extractFixCommand(outcome.value.content);
|
|
436
|
+
if (command === null) {
|
|
437
|
+
llmFailures += 1;
|
|
438
|
+
attempts.push(unexecuted(attempt, `completion carried no usable command: ${clipText(outcome.value.content, 300)}`, usage, null, started));
|
|
439
|
+
continue;
|
|
440
|
+
}
|
|
441
|
+
if (attempt > 1) {
|
|
442
|
+
const commandLines = countScriptCommands(command);
|
|
443
|
+
if (commandLines > scriptCap) {
|
|
444
|
+
llmFailures += 1;
|
|
445
|
+
attempts.push(unexecuted(attempt, `script exceeds ${scriptCap} command lines (${commandLines})`, usage, null, started));
|
|
446
|
+
options.onProgress?.(`fix-loop attempt ${attempt}: rejected script with ${commandLines} command lines`);
|
|
447
|
+
continue;
|
|
448
|
+
}
|
|
449
|
+
}
|
|
450
|
+
let execution;
|
|
451
|
+
try {
|
|
452
|
+
execution = await executor(command, attempt);
|
|
453
|
+
} catch (err) {
|
|
454
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
455
|
+
attempts.push({
|
|
456
|
+
...unexecuted(attempt, "", usage, command, started),
|
|
457
|
+
llmError: null,
|
|
458
|
+
armBError: message.slice(0, 500)
|
|
459
|
+
});
|
|
460
|
+
options.onProgress?.(`fix-loop attempt ${attempt}: sandbox error — ${message.slice(0, 160)}`);
|
|
461
|
+
return {
|
|
462
|
+
flipped: false,
|
|
463
|
+
flippedAtAttempt: null,
|
|
464
|
+
attempts,
|
|
465
|
+
aborted: true,
|
|
466
|
+
llmCalls,
|
|
467
|
+
llmFailures,
|
|
468
|
+
promptTokens,
|
|
469
|
+
completionTokens,
|
|
470
|
+
callsWithoutUsage
|
|
471
|
+
};
|
|
472
|
+
}
|
|
473
|
+
attempts.push({
|
|
474
|
+
attempt,
|
|
475
|
+
command,
|
|
476
|
+
llmError: null,
|
|
477
|
+
usage,
|
|
478
|
+
executed: true,
|
|
479
|
+
exitCode: execution.exitCode,
|
|
480
|
+
prefixExecuted: execution.prefixExecuted,
|
|
481
|
+
prefixDivergences: execution.prefixDivergences,
|
|
482
|
+
prefixDivergencePct: execution.prefixDivergencePct,
|
|
483
|
+
failureVanished: execution.failureVanished,
|
|
484
|
+
stdoutTail: clipText(execution.stdout, tailChars),
|
|
485
|
+
stderrTail: clipText(execution.stderr, tailChars),
|
|
486
|
+
armBError: null,
|
|
487
|
+
wallMs: Date.now() - started
|
|
488
|
+
});
|
|
489
|
+
options.onProgress?.(`fix-loop attempt ${attempt}: exit=${execution.exitCode} failureVanished=${execution.failureVanished}`);
|
|
490
|
+
if (execution.failureVanished) return {
|
|
491
|
+
flipped: true,
|
|
492
|
+
flippedAtAttempt: attempt,
|
|
493
|
+
attempts,
|
|
494
|
+
aborted: false,
|
|
495
|
+
llmCalls,
|
|
496
|
+
llmFailures,
|
|
497
|
+
promptTokens,
|
|
498
|
+
completionTokens,
|
|
499
|
+
callsWithoutUsage
|
|
500
|
+
};
|
|
501
|
+
}
|
|
502
|
+
return {
|
|
503
|
+
flipped: false,
|
|
504
|
+
flippedAtAttempt: null,
|
|
505
|
+
attempts,
|
|
506
|
+
aborted: false,
|
|
507
|
+
llmCalls,
|
|
508
|
+
llmFailures,
|
|
509
|
+
promptTokens,
|
|
510
|
+
completionTokens,
|
|
511
|
+
callsWithoutUsage
|
|
512
|
+
};
|
|
513
|
+
}
|
|
514
|
+
//#endregion
|
|
515
|
+
//#region src/trajectory-replay/image-preparer.ts
|
|
516
|
+
/**
|
|
517
|
+
* Replay-ready image derivation.
|
|
518
|
+
*
|
|
519
|
+
* A recorded trajectory names the image it ran in, but the image is not always
|
|
520
|
+
* runnable as-is: sandbox platforms pin customer commands to a non-root
|
|
521
|
+
* identity, so a root-owned working tree must be chowned before the replay can
|
|
522
|
+
* write to it. `ImagePreparer` is that step, injectable so a consumer whose
|
|
523
|
+
* images are already replay-ready supplies its own no-op or none at all.
|
|
524
|
+
*/
|
|
525
|
+
const execFileAsync = promisify(execFile);
|
|
526
|
+
function derivedImageTag(image, cwd) {
|
|
527
|
+
return `ctb-replay:${createHash("sha256").update(`${image}\n${cwd}`).digest("hex").slice(0, 12)}-uid1000`;
|
|
528
|
+
}
|
|
529
|
+
/**
|
|
530
|
+
* Pulls the base image when absent and builds `FROM <base>; RUN chown -R
|
|
531
|
+
* 1000:1000 <cwd>` tagged by content hash, so repeated batches reuse both
|
|
532
|
+
* the pull and the build. cwd `/` skips the chown (never chown -R /) and
|
|
533
|
+
* replays on the base image directly.
|
|
534
|
+
*/
|
|
535
|
+
function dockerImagePreparer(options = {}) {
|
|
536
|
+
const pullTimeoutMs = options.pullTimeoutMs ?? 12e5;
|
|
537
|
+
const buildTimeoutMs = options.buildTimeoutMs ?? 9e5;
|
|
538
|
+
const imageExists = async (tag) => {
|
|
539
|
+
try {
|
|
540
|
+
await execFileAsync("docker", [
|
|
541
|
+
"image",
|
|
542
|
+
"inspect",
|
|
543
|
+
tag
|
|
544
|
+
], { maxBuffer: 8 * 1024 * 1024 });
|
|
545
|
+
return true;
|
|
546
|
+
} catch {
|
|
547
|
+
return false;
|
|
548
|
+
}
|
|
549
|
+
};
|
|
550
|
+
const errorTail = (err) => {
|
|
551
|
+
return (err && typeof err === "object" && "stderr" in err && typeof err.stderr === "string" && err.stderr.length > 0 ? err.stderr : err instanceof Error ? err.message : String(err)).trim().split("\n").slice(-3).join(" | ").slice(0, 400);
|
|
552
|
+
};
|
|
553
|
+
return { async ensure(image, cwd) {
|
|
554
|
+
if (cwd === "/") {
|
|
555
|
+
if (await imageExists(image)) return {
|
|
556
|
+
succeeded: true,
|
|
557
|
+
value: {
|
|
558
|
+
derivedImage: image,
|
|
559
|
+
pulled: false,
|
|
560
|
+
built: false
|
|
561
|
+
}
|
|
562
|
+
};
|
|
563
|
+
try {
|
|
564
|
+
await execFileAsync("docker", ["pull", image], {
|
|
565
|
+
timeout: pullTimeoutMs,
|
|
566
|
+
maxBuffer: 32 * 1024 * 1024
|
|
567
|
+
});
|
|
568
|
+
} catch (err) {
|
|
569
|
+
return {
|
|
570
|
+
succeeded: false,
|
|
571
|
+
error: `pull ${image}: ${errorTail(err)}`
|
|
572
|
+
};
|
|
573
|
+
}
|
|
574
|
+
return {
|
|
575
|
+
succeeded: true,
|
|
576
|
+
value: {
|
|
577
|
+
derivedImage: image,
|
|
578
|
+
pulled: true,
|
|
579
|
+
built: false
|
|
580
|
+
}
|
|
581
|
+
};
|
|
582
|
+
}
|
|
583
|
+
const derived = derivedImageTag(image, cwd);
|
|
584
|
+
if (await imageExists(derived)) return {
|
|
585
|
+
succeeded: true,
|
|
586
|
+
value: {
|
|
587
|
+
derivedImage: derived,
|
|
588
|
+
pulled: false,
|
|
589
|
+
built: false
|
|
590
|
+
}
|
|
591
|
+
};
|
|
592
|
+
let pulled = false;
|
|
593
|
+
if (!await imageExists(image)) try {
|
|
594
|
+
await execFileAsync("docker", ["pull", image], {
|
|
595
|
+
timeout: pullTimeoutMs,
|
|
596
|
+
maxBuffer: 32 * 1024 * 1024
|
|
597
|
+
});
|
|
598
|
+
pulled = true;
|
|
599
|
+
} catch (err) {
|
|
600
|
+
return {
|
|
601
|
+
succeeded: false,
|
|
602
|
+
error: `pull ${image}: ${errorTail(err)}`
|
|
603
|
+
};
|
|
604
|
+
}
|
|
605
|
+
const contextDir = mkdtempSync(join(tmpdir(), "ctb-replay-image-"));
|
|
606
|
+
const quotedCwd = `'${cwd.replaceAll("'", `'\\''`)}'`;
|
|
607
|
+
writeFileSync(join(contextDir, "Dockerfile"), `FROM ${image}\nRUN chown -R 1000:1000 ${quotedCwd}\n`);
|
|
608
|
+
try {
|
|
609
|
+
await execFileAsync("docker", [
|
|
610
|
+
"build",
|
|
611
|
+
"-t",
|
|
612
|
+
derived,
|
|
613
|
+
contextDir
|
|
614
|
+
], {
|
|
615
|
+
timeout: buildTimeoutMs,
|
|
616
|
+
maxBuffer: 32 * 1024 * 1024
|
|
617
|
+
});
|
|
618
|
+
} catch (err) {
|
|
619
|
+
return {
|
|
620
|
+
succeeded: false,
|
|
621
|
+
error: `build ${derived} from ${image}: ${errorTail(err)}`
|
|
622
|
+
};
|
|
623
|
+
}
|
|
624
|
+
return {
|
|
625
|
+
succeeded: true,
|
|
626
|
+
value: {
|
|
627
|
+
derivedImage: derived,
|
|
628
|
+
pulled,
|
|
629
|
+
built: true
|
|
630
|
+
}
|
|
631
|
+
};
|
|
632
|
+
} };
|
|
633
|
+
}
|
|
634
|
+
//#endregion
|
|
635
|
+
//#region src/trajectory-replay/verify.ts
|
|
636
|
+
/**
|
|
637
|
+
* Execution-replay verification for recorded shell trajectories.
|
|
638
|
+
*
|
|
639
|
+
* Turns a cited claim ("step k is the error-critical step") into an executed
|
|
640
|
+
* proof: replay steps 1..k-1 inside the trajectory's own image, then
|
|
641
|
+
* arm A — run the recorded step k and check the recorded failure signature
|
|
642
|
+
* reproduces (returncode + a stable output substring), and
|
|
643
|
+
* arm B — run a corrected step k and check the failure vanishes.
|
|
644
|
+
*
|
|
645
|
+
* Built on the counterfactual scaffold: the trajectory is ingested into a
|
|
646
|
+
* TraceStore, each arm is a `runCounterfactual` meta-run, and the
|
|
647
|
+
* `SandboxCounterfactualRunner` here supplies the `executeFrom` callback that
|
|
648
|
+
* scaffold deliberately leaves to consumers.
|
|
649
|
+
*
|
|
650
|
+
* Honest limits:
|
|
651
|
+
* - Only trajectories that record their image can be replayed; tasks whose
|
|
652
|
+
* environment needs external compose peers cannot be replayed this way.
|
|
653
|
+
* - Each arm replays the prefix in its own fresh session, so a backend
|
|
654
|
+
* without copy-on-write forks pays the prefix twice.
|
|
655
|
+
* - Prefix divergence is recorded per step and surfaced in the verdict, never
|
|
656
|
+
* hidden: a high divergence rate is a finding about replayability, not an
|
|
657
|
+
* error of the harness.
|
|
658
|
+
*
|
|
659
|
+
* Agreement requires positive evidence. A prefix step counts as confirmed only
|
|
660
|
+
* when the recording carries a returncode AND the replayed exit equals it. A
|
|
661
|
+
* step the recording cannot adjudicate is a divergence of its own kind
|
|
662
|
+
* (`unknown-expectation`), never silent agreement — otherwise a replay that
|
|
663
|
+
* fails on every step reports a perfect prefix and every downstream verdict
|
|
664
|
+
* built on it is meaningless.
|
|
665
|
+
*/
|
|
666
|
+
/**
|
|
667
|
+
* Emit one tool span per trajectory step, in order, with a monotonic
|
|
668
|
+
* injected clock so `buildTrajectory` ordering is deterministic even when
|
|
669
|
+
* two spans would share a Date.now() millisecond.
|
|
670
|
+
*/
|
|
671
|
+
async function ingestRecordedTrajectory(store, steps, caseId) {
|
|
672
|
+
let tick = 0;
|
|
673
|
+
const emitter = new TraceEmitter(store, { now: () => ++tick });
|
|
674
|
+
await emitter.startRun({
|
|
675
|
+
scenarioId: caseId,
|
|
676
|
+
tags: {
|
|
677
|
+
source: "trajectory-replay",
|
|
678
|
+
caseId
|
|
679
|
+
}
|
|
680
|
+
});
|
|
681
|
+
for (const step of steps) {
|
|
682
|
+
if (typeof step.action !== "string" || step.action.length === 0) throw new Error(`trajectory-replay: step ${step.step_id} has no action`);
|
|
683
|
+
await (await emitter.tool({
|
|
684
|
+
name: `step:${step.step_id}`,
|
|
685
|
+
toolName: "shell",
|
|
686
|
+
args: { command: step.action },
|
|
687
|
+
result: {
|
|
688
|
+
returncode: parseRecordedReturncode(step.observation),
|
|
689
|
+
observation: step.observation
|
|
690
|
+
}
|
|
691
|
+
})).end();
|
|
692
|
+
}
|
|
693
|
+
await emitter.endRun({
|
|
694
|
+
pass: true,
|
|
695
|
+
notes: "recorded trajectory (ingested for replay)"
|
|
696
|
+
});
|
|
697
|
+
return {
|
|
698
|
+
runId: emitter.runId,
|
|
699
|
+
stepCount: steps.length
|
|
700
|
+
};
|
|
701
|
+
}
|
|
702
|
+
/** Admission tolerance: a prefix replay is faithful enough to build a verdict
|
|
703
|
+
* on when at most this percentage of its executed steps diverged. */
|
|
704
|
+
const PREFIX_DIVERGENCE_TOLERANCE_PCT = 10;
|
|
705
|
+
/**
|
|
706
|
+
* Compare one replayed prefix step against its recording. Returns null only
|
|
707
|
+
* when the recording positively confirms the replay.
|
|
708
|
+
*/
|
|
709
|
+
function classifyPrefixStep(step, expectedReturncode, actualExit) {
|
|
710
|
+
if (expectedReturncode === null) return {
|
|
711
|
+
step,
|
|
712
|
+
kind: "unknown-expectation",
|
|
713
|
+
expectedReturncode: null,
|
|
714
|
+
actualExit
|
|
715
|
+
};
|
|
716
|
+
if (expectedReturncode !== actualExit) return {
|
|
717
|
+
step,
|
|
718
|
+
kind: "returncode-mismatch",
|
|
719
|
+
expectedReturncode,
|
|
720
|
+
actualExit
|
|
721
|
+
};
|
|
722
|
+
return null;
|
|
723
|
+
}
|
|
724
|
+
/** Roll per-step classifications up into the rate the admission pre-pass gates on. */
|
|
725
|
+
function summarizePrefixReplay(prefixExecuted, prefixDivergences, wallMs) {
|
|
726
|
+
if (prefixDivergences.length > prefixExecuted) throw new Error(`trajectory-replay: ${prefixDivergences.length} divergences over ${prefixExecuted} executed prefix steps`);
|
|
727
|
+
const pct = prefixExecuted === 0 ? 0 : Number((prefixDivergences.length / prefixExecuted * 100).toFixed(1));
|
|
728
|
+
return {
|
|
729
|
+
prefixExecuted,
|
|
730
|
+
prefixDivergences,
|
|
731
|
+
prefixConfirmed: prefixExecuted - prefixDivergences.length,
|
|
732
|
+
prefixReturncodeMismatches: prefixDivergences.filter((d) => d.kind === "returncode-mismatch").length,
|
|
733
|
+
prefixUnknownExpectations: prefixDivergences.filter((d) => d.kind === "unknown-expectation").length,
|
|
734
|
+
prefixDivergencePct: pct,
|
|
735
|
+
prefixWithinTolerance: pct <= 10,
|
|
736
|
+
wallMs
|
|
737
|
+
};
|
|
738
|
+
}
|
|
739
|
+
function toolCommand(span) {
|
|
740
|
+
const args = span.args;
|
|
741
|
+
if (!args || typeof args.command !== "string") throw new Error(`trajectory-replay: span ${span.name} carries no command`);
|
|
742
|
+
return args.command;
|
|
743
|
+
}
|
|
744
|
+
function recordedReturncodeOf(span) {
|
|
745
|
+
const result = span.result;
|
|
746
|
+
return typeof result?.returncode === "number" ? result.returncode : null;
|
|
747
|
+
}
|
|
748
|
+
/**
|
|
749
|
+
* Replays `ctx.prefix` in a fresh session, then executes the mutated step.
|
|
750
|
+
* Divergences are recorded and never abort the replay. Results land on
|
|
751
|
+
* `lastPrefix` / `lastArm` for the caller; spans for every exec land in the
|
|
752
|
+
* counterfactual meta-run.
|
|
753
|
+
*/
|
|
754
|
+
var SandboxCounterfactualRunner = class {
|
|
755
|
+
backend;
|
|
756
|
+
options;
|
|
757
|
+
lastPrefix = null;
|
|
758
|
+
lastArm = null;
|
|
759
|
+
constructor(backend, options) {
|
|
760
|
+
this.backend = backend;
|
|
761
|
+
this.options = options;
|
|
762
|
+
}
|
|
763
|
+
async executeFrom(ctx, emitter) {
|
|
764
|
+
const { cwd, stepTimeoutMs, prefixLimit, onProgress } = this.options;
|
|
765
|
+
const session = await this.backend.open();
|
|
766
|
+
try {
|
|
767
|
+
const toolSteps = ctx.prefix.filter((s) => s.span.kind === "tool");
|
|
768
|
+
const toExecute = prefixLimit !== void 0 ? toolSteps.slice(0, prefixLimit) : toolSteps;
|
|
769
|
+
const divergences = [];
|
|
770
|
+
const prefixStart = Date.now();
|
|
771
|
+
for (const step of toExecute) {
|
|
772
|
+
const command = toolCommand(step.span);
|
|
773
|
+
const expected = recordedReturncodeOf(step.span);
|
|
774
|
+
const started = Date.now();
|
|
775
|
+
const result = await session.exec(wrapActionForExec(command, cwd), stepTimeoutMs);
|
|
776
|
+
await (await emitter.sandbox({
|
|
777
|
+
name: step.span.name,
|
|
778
|
+
command,
|
|
779
|
+
exitCode: result.exitCode,
|
|
780
|
+
wallMs: Date.now() - started
|
|
781
|
+
})).end();
|
|
782
|
+
const divergence = classifyPrefixStep(step.index + 1, expected, result.exitCode);
|
|
783
|
+
if (divergence) divergences.push(divergence);
|
|
784
|
+
onProgress?.(`prefix ${step.span.name}: exit=${result.exitCode}` + (divergence ? ` (${divergence.kind}${expected === null ? "" : `, recorded ${expected}`})` : ""));
|
|
785
|
+
}
|
|
786
|
+
this.lastPrefix = summarizePrefixReplay(toExecute.length, divergences, Date.now() - prefixStart);
|
|
787
|
+
if (ctx.mutatedStep.span.kind !== "tool") throw new Error("trajectory-replay: mutation target is not a tool span");
|
|
788
|
+
const armCommand = toolCommand(ctx.mutatedStep.span);
|
|
789
|
+
const armStart = Date.now();
|
|
790
|
+
const armResult = await session.exec(wrapActionForExec(armCommand, cwd), stepTimeoutMs);
|
|
791
|
+
const armWallMs = Date.now() - armStart;
|
|
792
|
+
await (await emitter.sandbox({
|
|
793
|
+
name: `candidate:${ctx.mutatedStep.span.name}`,
|
|
794
|
+
command: armCommand,
|
|
795
|
+
exitCode: armResult.exitCode,
|
|
796
|
+
wallMs: armWallMs
|
|
797
|
+
})).end();
|
|
798
|
+
this.lastArm = {
|
|
799
|
+
command: armCommand,
|
|
800
|
+
exitCode: armResult.exitCode,
|
|
801
|
+
stdout: armResult.stdout,
|
|
802
|
+
stderr: armResult.stderr,
|
|
803
|
+
wallMs: armWallMs
|
|
804
|
+
};
|
|
805
|
+
onProgress?.(`candidate step: exit=${armResult.exitCode} in ${armWallMs}ms`);
|
|
806
|
+
await emitter.endRun({
|
|
807
|
+
pass: armResult.exitCode === 0,
|
|
808
|
+
notes: `prefix ${toExecute.length} steps, ${divergences.length} divergences (${this.lastPrefix.prefixDivergencePct}%)`
|
|
809
|
+
});
|
|
810
|
+
} finally {
|
|
811
|
+
await session.close();
|
|
812
|
+
}
|
|
813
|
+
}
|
|
814
|
+
};
|
|
815
|
+
function excerpt(text, limit = 2400) {
|
|
816
|
+
if (text.length <= limit) return text;
|
|
817
|
+
const head = text.slice(0, limit / 2);
|
|
818
|
+
const tail = text.slice(-limit / 2);
|
|
819
|
+
return `${head}\n… [${text.length - limit} chars elided] …\n${tail}`;
|
|
820
|
+
}
|
|
821
|
+
async function replayVerify(options) {
|
|
822
|
+
const totalStart = Date.now();
|
|
823
|
+
const steps = JSON.parse(readFileSync(options.stepsPath, "utf8"));
|
|
824
|
+
if (!Array.isArray(steps) || steps.length === 0) throw new Error(`trajectory-replay: ${options.stepsPath} is not a non-empty steps array`);
|
|
825
|
+
const index = options.at - 1;
|
|
826
|
+
if (index < 0 || index >= steps.length) throw new Error(`trajectory-replay: at ${options.at} out of range [1, ${steps.length}]`);
|
|
827
|
+
const target = steps[index];
|
|
828
|
+
if (target.step_id !== options.at) throw new Error(`trajectory-replay: steps[${index}].step_id=${target.step_id} != at ${options.at}; steps.json must be ordered with 1-based contiguous step_ids`);
|
|
829
|
+
const caseId = options.caseId ?? options.stepsPath;
|
|
830
|
+
const recordedReturncode = parseRecordedReturncode(target.observation);
|
|
831
|
+
const signature = options.signature ?? deriveFailureSignature(target.observation);
|
|
832
|
+
const signatureBasis = signature ? "returncode+output-substring" : "returncode-only";
|
|
833
|
+
const runnerOptions = {
|
|
834
|
+
cwd: options.cwd,
|
|
835
|
+
stepTimeoutMs: options.stepTimeoutMs ?? 3e5,
|
|
836
|
+
prefixLimit: options.prefixLimit,
|
|
837
|
+
onProgress: options.onProgress
|
|
838
|
+
};
|
|
839
|
+
const store = new InMemoryTraceStore();
|
|
840
|
+
const { runId } = await ingestRecordedTrajectory(store, steps, caseId);
|
|
841
|
+
options.onProgress?.(`arm A: replaying ${index} prefix steps then recorded step ${options.at}`);
|
|
842
|
+
const armARunner = new SandboxCounterfactualRunner(options.backend, runnerOptions);
|
|
843
|
+
const armAStart = Date.now();
|
|
844
|
+
const armAResult = await runCounterfactual(store, runId, {
|
|
845
|
+
kind: "custom",
|
|
846
|
+
at: index,
|
|
847
|
+
describe: "arm-A identity replay",
|
|
848
|
+
apply: (s) => s
|
|
849
|
+
}, armARunner);
|
|
850
|
+
const armAMs = Date.now() - armAStart;
|
|
851
|
+
const armAExec = requireExec(armARunner, "arm A");
|
|
852
|
+
const armAOutput = `${armAExec.stdout}\n${armAExec.stderr}`;
|
|
853
|
+
const failureSignatureMatch = recordedReturncode !== null && armAExec.exitCode === recordedReturncode && (signature ? armAOutput.includes(signature) : true);
|
|
854
|
+
let armBRunner = null;
|
|
855
|
+
let armBExec = null;
|
|
856
|
+
let armBRunId = null;
|
|
857
|
+
let armBMs = null;
|
|
858
|
+
if (options.fixCommand) {
|
|
859
|
+
const fixCommand = options.fixCommand;
|
|
860
|
+
options.onProgress?.("arm B: replaying prefix then corrected step");
|
|
861
|
+
armBRunner = new SandboxCounterfactualRunner(options.backend, runnerOptions);
|
|
862
|
+
const armBStart = Date.now();
|
|
863
|
+
const armBResult = await runCounterfactual(store, runId, {
|
|
864
|
+
kind: "custom",
|
|
865
|
+
at: index,
|
|
866
|
+
describe: "arm-B corrected step",
|
|
867
|
+
apply: (step) => ({
|
|
868
|
+
...step,
|
|
869
|
+
span: {
|
|
870
|
+
...step.span,
|
|
871
|
+
args: { command: fixCommand }
|
|
872
|
+
}
|
|
873
|
+
})
|
|
874
|
+
}, armBRunner);
|
|
875
|
+
armBMs = Date.now() - armBStart;
|
|
876
|
+
armBRunId = armBResult.counterfactualRunId;
|
|
877
|
+
armBExec = requireExec(armBRunner, "arm B");
|
|
878
|
+
}
|
|
879
|
+
const armBOutput = armBExec ? `${armBExec.stdout}\n${armBExec.stderr}` : null;
|
|
880
|
+
const armAPrefix = requirePrefix(armARunner, "arm A");
|
|
881
|
+
const verdict = {
|
|
882
|
+
case: caseId,
|
|
883
|
+
image: options.image,
|
|
884
|
+
driver: options.driverLabel ?? "docker",
|
|
885
|
+
k: options.at,
|
|
886
|
+
cwd: options.cwd,
|
|
887
|
+
recordedReturncode,
|
|
888
|
+
signature,
|
|
889
|
+
signatureBasis,
|
|
890
|
+
prefixExecuted: armAPrefix.prefixExecuted,
|
|
891
|
+
prefixDivergences: armAPrefix.prefixDivergences,
|
|
892
|
+
prefixDivergencePct: armAPrefix.prefixDivergencePct,
|
|
893
|
+
prefixConfirmed: armAPrefix.prefixConfirmed,
|
|
894
|
+
prefixReturncodeMismatches: armAPrefix.prefixReturncodeMismatches,
|
|
895
|
+
prefixUnknownExpectations: armAPrefix.prefixUnknownExpectations,
|
|
896
|
+
prefixWithinTolerance: armAPrefix.prefixWithinTolerance,
|
|
897
|
+
armA: {
|
|
898
|
+
command: armAExec.command,
|
|
899
|
+
exitCode: armAExec.exitCode,
|
|
900
|
+
wallMs: armAExec.wallMs,
|
|
901
|
+
prefix: armAPrefix,
|
|
902
|
+
failureSignatureMatch
|
|
903
|
+
},
|
|
904
|
+
armB: armBExec && armBRunner ? {
|
|
905
|
+
command: armBExec.command,
|
|
906
|
+
exitCode: armBExec.exitCode,
|
|
907
|
+
wallMs: armBExec.wallMs,
|
|
908
|
+
prefix: requirePrefix(armBRunner, "arm B"),
|
|
909
|
+
failureVanished: armBExec.exitCode === 0 && (signature && armBOutput ? !armBOutput.includes(signature) : true)
|
|
910
|
+
} : null,
|
|
911
|
+
timings: {
|
|
912
|
+
armAMs,
|
|
913
|
+
armBMs,
|
|
914
|
+
totalMs: Date.now() - totalStart
|
|
915
|
+
},
|
|
916
|
+
runIds: {
|
|
917
|
+
original: runId,
|
|
918
|
+
armA: armAResult.counterfactualRunId,
|
|
919
|
+
armB: armBRunId
|
|
920
|
+
}
|
|
921
|
+
};
|
|
922
|
+
mkdirSync(options.out, { recursive: true });
|
|
923
|
+
writeFileSync(join(options.out, "replay-verdict.json"), `${JSON.stringify(verdict, null, 2)}\n`);
|
|
924
|
+
writeFileSync(join(options.out, "report.md"), renderReport(verdict, target, armAExec, armBExec));
|
|
925
|
+
return verdict;
|
|
926
|
+
}
|
|
927
|
+
function requireExec(runner, arm) {
|
|
928
|
+
if (!runner.lastArm) throw new Error(`trajectory-replay: ${arm} finished without executing step k`);
|
|
929
|
+
return runner.lastArm;
|
|
930
|
+
}
|
|
931
|
+
function requirePrefix(runner, arm) {
|
|
932
|
+
if (!runner.lastPrefix) throw new Error(`trajectory-replay: ${arm} reported no prefix replay result`);
|
|
933
|
+
return runner.lastPrefix;
|
|
934
|
+
}
|
|
935
|
+
function renderReport(verdict, target, armA, armB) {
|
|
936
|
+
const lines = [];
|
|
937
|
+
lines.push(`# Replay verification — ${verdict.case}`);
|
|
938
|
+
lines.push("");
|
|
939
|
+
lines.push(`- image: \`${verdict.image}\` (driver: ${verdict.driver})`);
|
|
940
|
+
lines.push(`- error-critical step k = ${verdict.k}, cwd \`${verdict.cwd}\``);
|
|
941
|
+
lines.push(`- recorded returncode at k: ${verdict.recordedReturncode ?? "none recorded"}`);
|
|
942
|
+
lines.push(`- failure signature (${verdict.signatureBasis}): ${verdict.signature ? `\`${verdict.signature}\`` : "—"}`);
|
|
943
|
+
lines.push("");
|
|
944
|
+
lines.push(`## Prefix replay (steps 1..${verdict.k - 1})`);
|
|
945
|
+
lines.push("");
|
|
946
|
+
lines.push(`Executed ${verdict.prefixExecuted} steps; ${verdict.prefixConfirmed} confirmed against the recording, ${verdict.prefixDivergences.length} divergent (${verdict.prefixDivergencePct}%) — ${verdict.prefixReturncodeMismatches} returncode mismatches, ${verdict.prefixUnknownExpectations} with no recorded returncode to check.`);
|
|
947
|
+
lines.push("");
|
|
948
|
+
lines.push(`Within the 10% admission tolerance: **${verdict.prefixWithinTolerance}**.`);
|
|
949
|
+
if (verdict.prefixDivergences.length > 0) {
|
|
950
|
+
lines.push("");
|
|
951
|
+
lines.push("| step | kind | recorded rc | replayed exit |");
|
|
952
|
+
lines.push("| --- | --- | --- | --- |");
|
|
953
|
+
for (const d of verdict.prefixDivergences) lines.push(`| ${d.step} | ${d.kind} | ${d.expectedReturncode ?? "none recorded"} | ${d.actualExit} |`);
|
|
954
|
+
}
|
|
955
|
+
lines.push("");
|
|
956
|
+
lines.push("## Arm A — recorded step k, replayed");
|
|
957
|
+
lines.push("");
|
|
958
|
+
lines.push("```sh");
|
|
959
|
+
lines.push(armA.command);
|
|
960
|
+
lines.push("```");
|
|
961
|
+
lines.push("");
|
|
962
|
+
lines.push(`Exit ${armA.exitCode} in ${armA.wallMs}ms — failureSignatureMatch: **${verdict.armA.failureSignatureMatch}**`);
|
|
963
|
+
lines.push("");
|
|
964
|
+
lines.push("Recorded observation excerpt:");
|
|
965
|
+
lines.push("");
|
|
966
|
+
lines.push("```");
|
|
967
|
+
lines.push(excerpt(parseObservationOutput(target.observation)));
|
|
968
|
+
lines.push("```");
|
|
969
|
+
lines.push("");
|
|
970
|
+
lines.push("Replayed stdout+stderr excerpt:");
|
|
971
|
+
lines.push("");
|
|
972
|
+
lines.push("```");
|
|
973
|
+
lines.push(excerpt(`${armA.stdout}\n${armA.stderr}`.trim()));
|
|
974
|
+
lines.push("```");
|
|
975
|
+
if (armB && verdict.armB) {
|
|
976
|
+
lines.push("");
|
|
977
|
+
lines.push("## Arm B — corrected step k");
|
|
978
|
+
lines.push("");
|
|
979
|
+
lines.push("```sh");
|
|
980
|
+
lines.push(armB.command);
|
|
981
|
+
lines.push("```");
|
|
982
|
+
lines.push("");
|
|
983
|
+
lines.push(`Exit ${armB.exitCode} in ${armB.wallMs}ms — failureVanished: **${verdict.armB.failureVanished}** (prefix re-replayed in a fresh session: ${verdict.armB.prefix.prefixExecuted} steps, ${verdict.armB.prefix.prefixDivergences.length} divergences, ${verdict.armB.prefix.prefixDivergencePct}%)`);
|
|
984
|
+
lines.push("");
|
|
985
|
+
lines.push("Replayed stdout+stderr excerpt:");
|
|
986
|
+
lines.push("");
|
|
987
|
+
lines.push("```");
|
|
988
|
+
lines.push(excerpt(`${armB.stdout}\n${armB.stderr}`.trim()));
|
|
989
|
+
lines.push("```");
|
|
990
|
+
}
|
|
991
|
+
lines.push("");
|
|
992
|
+
lines.push("## Timings");
|
|
993
|
+
lines.push("");
|
|
994
|
+
lines.push(`arm A ${verdict.timings.armAMs}ms, arm B ${verdict.timings.armBMs ?? "—"}ms, total ${verdict.timings.totalMs}ms.`);
|
|
995
|
+
lines.push("");
|
|
996
|
+
return lines.join("\n");
|
|
997
|
+
}
|
|
998
|
+
//#endregion
|
|
999
|
+
//#region src/trajectory-replay/batch.ts
|
|
1000
|
+
/**
|
|
1001
|
+
* Batch replay verification across gold-labeled trajectory corpora.
|
|
1002
|
+
*
|
|
1003
|
+
* For every replayable case (a raw trajectory with a recorded image AND at
|
|
1004
|
+
* least one gold incorrect step) the batch:
|
|
1005
|
+
* 1. derives the replay-ready image through the injected `ImagePreparer`,
|
|
1006
|
+
* 2. replays the prefix and runs arm A — the recorded gold step k, and
|
|
1007
|
+
* 3. optionally generates a corrected command with one LLM call and runs
|
|
1008
|
+
* arm B in its own fresh session.
|
|
1009
|
+
*
|
|
1010
|
+
* Headline metrics:
|
|
1011
|
+
* replayability rate — fraction of replayable cases where the prefix
|
|
1012
|
+
* replays within the divergence tolerance AND arm A reproduces the
|
|
1013
|
+
* recorded returncode at k;
|
|
1014
|
+
* prefix fidelity — executed prefix steps and the share of them that did
|
|
1015
|
+
* not confirm the recording, split by kind. A corpus whose recordings
|
|
1016
|
+
* carry no returncodes shows up here as unknown-expectation steps, never
|
|
1017
|
+
* as a clean replay;
|
|
1018
|
+
* fix-flip rate — fraction of arm-B-executed cases where the failure
|
|
1019
|
+
* vanished (exit 0, signature absent).
|
|
1020
|
+
*
|
|
1021
|
+
* Image pulls and execs run strictly serially: pulls contend on disk and
|
|
1022
|
+
* registry bandwidth, and serial cases keep wall-time attribution per case
|
|
1023
|
+
* honest. Pull failures are report rows, never silent skips.
|
|
1024
|
+
*/
|
|
1025
|
+
/** Deterministic PRNG for the fix-case sample; the seed lands in the report. */
|
|
1026
|
+
function mulberry32(seed) {
|
|
1027
|
+
let state = seed >>> 0;
|
|
1028
|
+
return () => {
|
|
1029
|
+
state = state + 1831565813 >>> 0;
|
|
1030
|
+
let t = state;
|
|
1031
|
+
t = Math.imul(t ^ t >>> 15, t | 1);
|
|
1032
|
+
t ^= t + Math.imul(t ^ t >>> 7, t | 61);
|
|
1033
|
+
return ((t ^ t >>> 14) >>> 0) / 4294967296;
|
|
1034
|
+
};
|
|
1035
|
+
}
|
|
1036
|
+
function seededSample(items, size, seed) {
|
|
1037
|
+
const pool = [...items];
|
|
1038
|
+
const random = mulberry32(seed);
|
|
1039
|
+
for (let i = pool.length - 1; i > 0; i--) {
|
|
1040
|
+
const j = Math.floor(random() * (i + 1));
|
|
1041
|
+
[pool[i], pool[j]] = [pool[j], pool[i]];
|
|
1042
|
+
}
|
|
1043
|
+
return new Set(pool.slice(0, size));
|
|
1044
|
+
}
|
|
1045
|
+
function caseOutDirName(row) {
|
|
1046
|
+
return `${row.corpus}--${row.trajId}`.replaceAll(/[^A-Za-z0-9._-]/g, "_").slice(0, 180);
|
|
1047
|
+
}
|
|
1048
|
+
/**
|
|
1049
|
+
* Arm B standalone: fresh sandbox, prefix replay, corrected step k. Reuses
|
|
1050
|
+
* the same counterfactual scaffold as replayVerify without re-running arm A.
|
|
1051
|
+
*/
|
|
1052
|
+
async function executeArmB(replayCase, fixCommand, backend, signature, stepTimeoutMs, prefixLimit, onProgress) {
|
|
1053
|
+
const store = new InMemoryTraceStore();
|
|
1054
|
+
const { runId } = await ingestRecordedTrajectory(store, replayCase.steps, replayCase.trajId);
|
|
1055
|
+
const runner = new SandboxCounterfactualRunner(backend, {
|
|
1056
|
+
cwd: replayCase.cwd,
|
|
1057
|
+
stepTimeoutMs,
|
|
1058
|
+
prefixLimit,
|
|
1059
|
+
onProgress
|
|
1060
|
+
});
|
|
1061
|
+
await runCounterfactual(store, runId, {
|
|
1062
|
+
kind: "custom",
|
|
1063
|
+
at: replayCase.k - 1,
|
|
1064
|
+
describe: "arm-B corrected step",
|
|
1065
|
+
apply: (step) => ({
|
|
1066
|
+
...step,
|
|
1067
|
+
span: {
|
|
1068
|
+
...step.span,
|
|
1069
|
+
args: { command: fixCommand }
|
|
1070
|
+
}
|
|
1071
|
+
})
|
|
1072
|
+
}, runner);
|
|
1073
|
+
const exec = runner.lastArm;
|
|
1074
|
+
if (!exec) throw new Error("arm B finished without executing the corrected step");
|
|
1075
|
+
const prefix = runner.lastPrefix;
|
|
1076
|
+
if (!prefix) throw new Error("arm B reported no prefix replay result");
|
|
1077
|
+
const output = `${exec.stdout}\n${exec.stderr}`;
|
|
1078
|
+
return {
|
|
1079
|
+
exitCode: exec.exitCode,
|
|
1080
|
+
prefixExecuted: prefix.prefixExecuted,
|
|
1081
|
+
prefixDivergences: prefix.prefixDivergences.length,
|
|
1082
|
+
prefixDivergencePct: prefix.prefixDivergencePct,
|
|
1083
|
+
failureVanished: exec.exitCode === 0 && (signature ? !output.includes(signature) : true),
|
|
1084
|
+
stdout: exec.stdout,
|
|
1085
|
+
stderr: exec.stderr
|
|
1086
|
+
};
|
|
1087
|
+
}
|
|
1088
|
+
function rate(numerator, denominator) {
|
|
1089
|
+
return denominator > 0 ? numerator / denominator : null;
|
|
1090
|
+
}
|
|
1091
|
+
async function runReplayBatch(options) {
|
|
1092
|
+
const onProgress = options.onProgress ?? (() => {});
|
|
1093
|
+
const enumeration = enumerateReplayableCases(options.corpora);
|
|
1094
|
+
let selected = enumeration.replayable;
|
|
1095
|
+
if (options.caseFilter) selected = selected.filter((c) => c.trajId.includes(options.caseFilter));
|
|
1096
|
+
if (options.caseLimit !== void 0) selected = selected.slice(0, options.caseLimit);
|
|
1097
|
+
onProgress(`enumerated ${enumeration.labelEntryCount} label entries → ${enumeration.replayable.length} replayable, ${enumeration.excluded.length} excluded; executing ${selected.length}`);
|
|
1098
|
+
const preparer = options.preparer ?? dockerImagePreparer();
|
|
1099
|
+
const backendFactory = options.backendFactory;
|
|
1100
|
+
mkdirSync(options.out, { recursive: true });
|
|
1101
|
+
const progressPath = join(options.out, "cases.jsonl");
|
|
1102
|
+
const rows = [];
|
|
1103
|
+
const pullFailures = [];
|
|
1104
|
+
const verdictByTraj = /* @__PURE__ */ new Map();
|
|
1105
|
+
for (const [index, replayCase] of selected.entries()) {
|
|
1106
|
+
const caseStart = Date.now();
|
|
1107
|
+
const label = `[${index + 1}/${selected.length}] ${replayCase.corpus}/${replayCase.trajId}`;
|
|
1108
|
+
onProgress(`${label}: preparing image ${replayCase.image}`);
|
|
1109
|
+
const preparation = await preparer.ensure(replayCase.image, replayCase.cwd);
|
|
1110
|
+
const base = {
|
|
1111
|
+
corpus: replayCase.corpus,
|
|
1112
|
+
trajId: replayCase.trajId,
|
|
1113
|
+
image: replayCase.image,
|
|
1114
|
+
cwd: replayCase.cwd,
|
|
1115
|
+
cwdSource: replayCase.cwdSource,
|
|
1116
|
+
k: replayCase.k,
|
|
1117
|
+
stepCount: replayCase.steps.length,
|
|
1118
|
+
goldIncorrectSteps: replayCase.goldIncorrectSteps,
|
|
1119
|
+
submitGoldsSkipped: replayCase.submitGoldsSkipped,
|
|
1120
|
+
recordedReturncodeAtK: replayCase.recordedReturncodeAtK
|
|
1121
|
+
};
|
|
1122
|
+
if (!preparation.succeeded) {
|
|
1123
|
+
pullFailures.push({
|
|
1124
|
+
corpus: replayCase.corpus,
|
|
1125
|
+
trajId: replayCase.trajId,
|
|
1126
|
+
image: replayCase.image,
|
|
1127
|
+
error: preparation.error
|
|
1128
|
+
});
|
|
1129
|
+
const row = {
|
|
1130
|
+
...base,
|
|
1131
|
+
derivedImage: null,
|
|
1132
|
+
signature: null,
|
|
1133
|
+
status: "image-unavailable",
|
|
1134
|
+
error: preparation.error,
|
|
1135
|
+
imagePulled: false,
|
|
1136
|
+
imageBuilt: false,
|
|
1137
|
+
prefixExecuted: null,
|
|
1138
|
+
prefixDivergences: null,
|
|
1139
|
+
prefixDivergencePct: null,
|
|
1140
|
+
prefixConfirmed: null,
|
|
1141
|
+
prefixReturncodeMismatches: null,
|
|
1142
|
+
prefixUnknownExpectations: null,
|
|
1143
|
+
armAExit: null,
|
|
1144
|
+
armAReturncodeMatch: false,
|
|
1145
|
+
armASignatureMatch: false,
|
|
1146
|
+
replayed: false,
|
|
1147
|
+
fix: null,
|
|
1148
|
+
wallMs: Date.now() - caseStart
|
|
1149
|
+
};
|
|
1150
|
+
rows.push(row);
|
|
1151
|
+
appendFileSync(progressPath, `${JSON.stringify(row)}\n`);
|
|
1152
|
+
onProgress(`${label}: image unavailable — ${preparation.error}`);
|
|
1153
|
+
continue;
|
|
1154
|
+
}
|
|
1155
|
+
const { derivedImage, pulled, built } = preparation.value;
|
|
1156
|
+
const stepTimeoutMs = options.stepTimeoutMs ?? replayCase.recordedStepTimeoutMs ?? 12e4;
|
|
1157
|
+
const caseOut = join(options.out, caseOutDirName(replayCase));
|
|
1158
|
+
onProgress(`${label}: arm A on ${derivedImage} (k=${replayCase.k}, timeout ${stepTimeoutMs}ms)`);
|
|
1159
|
+
try {
|
|
1160
|
+
const verdict = await replayVerify({
|
|
1161
|
+
stepsPath: replayCase.stepsPath,
|
|
1162
|
+
image: derivedImage,
|
|
1163
|
+
at: replayCase.k,
|
|
1164
|
+
cwd: replayCase.cwd,
|
|
1165
|
+
out: caseOut,
|
|
1166
|
+
caseId: replayCase.trajId,
|
|
1167
|
+
stepTimeoutMs,
|
|
1168
|
+
prefixLimit: options.prefixLimit,
|
|
1169
|
+
backend: backendFactory(derivedImage),
|
|
1170
|
+
onProgress: (message) => onProgress(`${label}: ${message}`)
|
|
1171
|
+
});
|
|
1172
|
+
verdictByTraj.set(replayCase.trajId, verdict);
|
|
1173
|
+
const returncodeMatch = verdict.recordedReturncode !== null && verdict.armA.exitCode === verdict.recordedReturncode;
|
|
1174
|
+
const row = {
|
|
1175
|
+
...base,
|
|
1176
|
+
derivedImage,
|
|
1177
|
+
signature: verdict.signature,
|
|
1178
|
+
status: "ok",
|
|
1179
|
+
error: null,
|
|
1180
|
+
imagePulled: pulled,
|
|
1181
|
+
imageBuilt: built,
|
|
1182
|
+
prefixExecuted: verdict.prefixExecuted,
|
|
1183
|
+
prefixDivergences: verdict.prefixDivergences.length,
|
|
1184
|
+
prefixDivergencePct: verdict.prefixDivergencePct,
|
|
1185
|
+
prefixConfirmed: verdict.prefixConfirmed,
|
|
1186
|
+
prefixReturncodeMismatches: verdict.prefixReturncodeMismatches,
|
|
1187
|
+
prefixUnknownExpectations: verdict.prefixUnknownExpectations,
|
|
1188
|
+
armAExit: verdict.armA.exitCode,
|
|
1189
|
+
armAReturncodeMatch: returncodeMatch,
|
|
1190
|
+
armASignatureMatch: verdict.armA.failureSignatureMatch,
|
|
1191
|
+
replayed: verdict.prefixWithinTolerance && returncodeMatch,
|
|
1192
|
+
fix: null,
|
|
1193
|
+
wallMs: Date.now() - caseStart
|
|
1194
|
+
};
|
|
1195
|
+
rows.push(row);
|
|
1196
|
+
appendFileSync(progressPath, `${JSON.stringify(row)}\n`);
|
|
1197
|
+
onProgress(`${label}: armA exit=${verdict.armA.exitCode} rcMatch=${returncodeMatch} divergences=${verdict.prefixDivergences.length}/${verdict.prefixExecuted} (${verdict.prefixDivergencePct}%, ${verdict.prefixUnknownExpectations} unknown-expectation)`);
|
|
1198
|
+
} catch (err) {
|
|
1199
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
1200
|
+
const row = {
|
|
1201
|
+
...base,
|
|
1202
|
+
derivedImage,
|
|
1203
|
+
signature: null,
|
|
1204
|
+
status: "replay-error",
|
|
1205
|
+
error: message.slice(0, 500),
|
|
1206
|
+
imagePulled: pulled,
|
|
1207
|
+
imageBuilt: built,
|
|
1208
|
+
prefixExecuted: null,
|
|
1209
|
+
prefixDivergences: null,
|
|
1210
|
+
prefixDivergencePct: null,
|
|
1211
|
+
prefixConfirmed: null,
|
|
1212
|
+
prefixReturncodeMismatches: null,
|
|
1213
|
+
prefixUnknownExpectations: null,
|
|
1214
|
+
armAExit: null,
|
|
1215
|
+
armAReturncodeMatch: false,
|
|
1216
|
+
armASignatureMatch: false,
|
|
1217
|
+
replayed: false,
|
|
1218
|
+
fix: null,
|
|
1219
|
+
wallMs: Date.now() - caseStart
|
|
1220
|
+
};
|
|
1221
|
+
rows.push(row);
|
|
1222
|
+
appendFileSync(progressPath, `${JSON.stringify(row)}\n`);
|
|
1223
|
+
onProgress(`${label}: replay error — ${message.slice(0, 200)}`);
|
|
1224
|
+
}
|
|
1225
|
+
}
|
|
1226
|
+
let llm = null;
|
|
1227
|
+
if (options.fix !== "none") {
|
|
1228
|
+
const caller = options.fixCaller;
|
|
1229
|
+
if (!caller) throw new Error(`trajectory-replay: fix=${options.fix} requires a fixCaller`);
|
|
1230
|
+
const fixAttempts = options.fixAttempts ?? 3;
|
|
1231
|
+
const maxFixCases = options.maxFixCases ?? 30;
|
|
1232
|
+
const seed = options.seed ?? 17;
|
|
1233
|
+
const eligible = rows.filter((r) => r.status === "ok" && r.replayed);
|
|
1234
|
+
const sampled = eligible.length > maxFixCases ? seededSample(eligible, maxFixCases, seed) : new Set(eligible);
|
|
1235
|
+
if (eligible.length > maxFixCases) onProgress(`fix phase: ${eligible.length} eligible > cap ${maxFixCases}; seeded sample (seed=${seed})`);
|
|
1236
|
+
let calls = 0;
|
|
1237
|
+
let failures = 0;
|
|
1238
|
+
let promptTokens = 0;
|
|
1239
|
+
let completionTokens = 0;
|
|
1240
|
+
let callsWithoutUsage = 0;
|
|
1241
|
+
for (const row of eligible) {
|
|
1242
|
+
const index = rows.indexOf(row);
|
|
1243
|
+
if (!sampled.has(row)) {
|
|
1244
|
+
rows[index] = {
|
|
1245
|
+
...row,
|
|
1246
|
+
fix: {
|
|
1247
|
+
attempted: false,
|
|
1248
|
+
sampledOut: true,
|
|
1249
|
+
command: null,
|
|
1250
|
+
llmError: null,
|
|
1251
|
+
usage: null,
|
|
1252
|
+
armBExit: null,
|
|
1253
|
+
armBPrefixExecuted: null,
|
|
1254
|
+
armBPrefixDivergences: null,
|
|
1255
|
+
armBPrefixDivergencePct: null,
|
|
1256
|
+
failureVanished: null,
|
|
1257
|
+
armBError: null,
|
|
1258
|
+
attempts: null,
|
|
1259
|
+
flippedAtAttempt: null
|
|
1260
|
+
}
|
|
1261
|
+
};
|
|
1262
|
+
continue;
|
|
1263
|
+
}
|
|
1264
|
+
const replayCase = selected.find((c) => c.trajId === row.trajId && c.corpus === row.corpus);
|
|
1265
|
+
const label = `fix ${row.corpus}/${row.trajId}`;
|
|
1266
|
+
const signature = verdictByTraj.get(row.trajId)?.signature ?? null;
|
|
1267
|
+
const stepTimeoutMs = options.stepTimeoutMs ?? replayCase.recordedStepTimeoutMs ?? 12e4;
|
|
1268
|
+
const promptInput = {
|
|
1269
|
+
taskStatement: replayCase.taskStatement,
|
|
1270
|
+
steps: replayCase.steps,
|
|
1271
|
+
k: replayCase.k
|
|
1272
|
+
};
|
|
1273
|
+
const caseOut = join(options.out, caseOutDirName(row));
|
|
1274
|
+
if (options.fix === "loop") {
|
|
1275
|
+
onProgress(`${label}: fix loop (budget ${fixAttempts} attempts)`);
|
|
1276
|
+
const result = await runFixLoop(caller, promptInput, async (command, attempt) => {
|
|
1277
|
+
onProgress(`${label}: attempt ${attempt} arm B — ${command.split("\n")[0].slice(0, 120)}`);
|
|
1278
|
+
const armB = await executeArmB(replayCase, command, backendFactory(row.derivedImage), signature, stepTimeoutMs, options.prefixLimit, (message) => onProgress(`${label}: attempt ${attempt}: ${message}`));
|
|
1279
|
+
mkdirSync(caseOut, { recursive: true });
|
|
1280
|
+
writeFileSync(join(caseOut, `armB-attempt${attempt}-result.json`), `${JSON.stringify({
|
|
1281
|
+
command,
|
|
1282
|
+
...armB
|
|
1283
|
+
}, null, 2)}\n`);
|
|
1284
|
+
writeFileSync(join(caseOut, "armB-result.json"), `${JSON.stringify({
|
|
1285
|
+
command,
|
|
1286
|
+
attempt,
|
|
1287
|
+
...armB
|
|
1288
|
+
}, null, 2)}\n`);
|
|
1289
|
+
return armB;
|
|
1290
|
+
}, {
|
|
1291
|
+
maxAttempts: fixAttempts,
|
|
1292
|
+
onProgress: (message) => onProgress(`${label}: ${message}`)
|
|
1293
|
+
});
|
|
1294
|
+
calls += result.llmCalls;
|
|
1295
|
+
failures += result.llmFailures;
|
|
1296
|
+
promptTokens += result.promptTokens;
|
|
1297
|
+
completionTokens += result.completionTokens;
|
|
1298
|
+
callsWithoutUsage += result.callsWithoutUsage;
|
|
1299
|
+
const usage = result.promptTokens + result.completionTokens > 0 ? {
|
|
1300
|
+
promptTokens: result.promptTokens,
|
|
1301
|
+
completionTokens: result.completionTokens
|
|
1302
|
+
} : null;
|
|
1303
|
+
const summary = (result.flippedAtAttempt !== null ? result.attempts.find((a) => a.attempt === result.flippedAtAttempt) : null) ?? [...result.attempts].reverse().find((a) => a.executed) ?? null;
|
|
1304
|
+
const lastRecord = result.attempts.at(-1) ?? null;
|
|
1305
|
+
rows[index] = {
|
|
1306
|
+
...row,
|
|
1307
|
+
fix: summary ? {
|
|
1308
|
+
attempted: true,
|
|
1309
|
+
sampledOut: false,
|
|
1310
|
+
command: summary.command,
|
|
1311
|
+
llmError: null,
|
|
1312
|
+
usage,
|
|
1313
|
+
armBExit: summary.exitCode,
|
|
1314
|
+
armBPrefixExecuted: summary.prefixExecuted,
|
|
1315
|
+
armBPrefixDivergences: summary.prefixDivergences,
|
|
1316
|
+
armBPrefixDivergencePct: summary.prefixDivergencePct,
|
|
1317
|
+
failureVanished: summary.failureVanished,
|
|
1318
|
+
armBError: null,
|
|
1319
|
+
attempts: result.attempts,
|
|
1320
|
+
flippedAtAttempt: result.flippedAtAttempt
|
|
1321
|
+
} : result.aborted ? {
|
|
1322
|
+
attempted: true,
|
|
1323
|
+
sampledOut: false,
|
|
1324
|
+
command: lastRecord?.command ?? null,
|
|
1325
|
+
llmError: null,
|
|
1326
|
+
usage,
|
|
1327
|
+
armBExit: null,
|
|
1328
|
+
armBPrefixExecuted: null,
|
|
1329
|
+
armBPrefixDivergences: null,
|
|
1330
|
+
armBPrefixDivergencePct: null,
|
|
1331
|
+
failureVanished: null,
|
|
1332
|
+
armBError: lastRecord?.armBError ?? "sandbox error",
|
|
1333
|
+
attempts: result.attempts,
|
|
1334
|
+
flippedAtAttempt: null
|
|
1335
|
+
} : {
|
|
1336
|
+
attempted: true,
|
|
1337
|
+
sampledOut: false,
|
|
1338
|
+
command: null,
|
|
1339
|
+
llmError: lastRecord?.llmError ?? "no attempt produced a runnable fix",
|
|
1340
|
+
usage,
|
|
1341
|
+
armBExit: null,
|
|
1342
|
+
armBPrefixExecuted: null,
|
|
1343
|
+
armBPrefixDivergences: null,
|
|
1344
|
+
armBPrefixDivergencePct: null,
|
|
1345
|
+
failureVanished: null,
|
|
1346
|
+
armBError: null,
|
|
1347
|
+
attempts: result.attempts,
|
|
1348
|
+
flippedAtAttempt: null
|
|
1349
|
+
}
|
|
1350
|
+
};
|
|
1351
|
+
onProgress(`${label}: loop done — flipped=${result.flipped}` + (result.flippedAtAttempt !== null ? ` at attempt ${result.flippedAtAttempt}` : "") + ` (${result.llmCalls} calls, ${result.attempts.filter((a) => a.executed).length} arms)`);
|
|
1352
|
+
continue;
|
|
1353
|
+
}
|
|
1354
|
+
onProgress(`${label}: generating corrected command`);
|
|
1355
|
+
calls += 1;
|
|
1356
|
+
const generated = await generateFixCommand(caller, promptInput);
|
|
1357
|
+
if (!generated.succeeded) {
|
|
1358
|
+
failures += 1;
|
|
1359
|
+
rows[index] = {
|
|
1360
|
+
...row,
|
|
1361
|
+
fix: {
|
|
1362
|
+
attempted: true,
|
|
1363
|
+
sampledOut: false,
|
|
1364
|
+
command: null,
|
|
1365
|
+
llmError: generated.error,
|
|
1366
|
+
usage: null,
|
|
1367
|
+
armBExit: null,
|
|
1368
|
+
armBPrefixExecuted: null,
|
|
1369
|
+
armBPrefixDivergences: null,
|
|
1370
|
+
armBPrefixDivergencePct: null,
|
|
1371
|
+
failureVanished: null,
|
|
1372
|
+
armBError: null,
|
|
1373
|
+
attempts: null,
|
|
1374
|
+
flippedAtAttempt: null
|
|
1375
|
+
}
|
|
1376
|
+
};
|
|
1377
|
+
onProgress(`${label}: LLM failed — ${generated.error.slice(0, 200)}`);
|
|
1378
|
+
continue;
|
|
1379
|
+
}
|
|
1380
|
+
if (generated.value.usage === null || generated.value.usage === void 0) callsWithoutUsage += 1;
|
|
1381
|
+
promptTokens += generated.value.usage?.promptTokens ?? 0;
|
|
1382
|
+
completionTokens += generated.value.usage?.completionTokens ?? 0;
|
|
1383
|
+
onProgress(`${label}: arm B — ${generated.value.command.split("\n")[0].slice(0, 120)}`);
|
|
1384
|
+
try {
|
|
1385
|
+
const armB = await executeArmB(replayCase, generated.value.command, backendFactory(row.derivedImage), signature, stepTimeoutMs, options.prefixLimit, (message) => onProgress(`${label}: ${message}`));
|
|
1386
|
+
rows[index] = {
|
|
1387
|
+
...row,
|
|
1388
|
+
fix: {
|
|
1389
|
+
attempted: true,
|
|
1390
|
+
sampledOut: false,
|
|
1391
|
+
command: generated.value.command,
|
|
1392
|
+
llmError: null,
|
|
1393
|
+
usage: generated.value.usage,
|
|
1394
|
+
armBExit: armB.exitCode,
|
|
1395
|
+
armBPrefixExecuted: armB.prefixExecuted,
|
|
1396
|
+
armBPrefixDivergences: armB.prefixDivergences,
|
|
1397
|
+
armBPrefixDivergencePct: armB.prefixDivergencePct,
|
|
1398
|
+
failureVanished: armB.failureVanished,
|
|
1399
|
+
armBError: null,
|
|
1400
|
+
attempts: null,
|
|
1401
|
+
flippedAtAttempt: armB.failureVanished ? 1 : null
|
|
1402
|
+
}
|
|
1403
|
+
};
|
|
1404
|
+
mkdirSync(caseOut, { recursive: true });
|
|
1405
|
+
writeFileSync(join(caseOut, "armB-result.json"), `${JSON.stringify({
|
|
1406
|
+
command: generated.value.command,
|
|
1407
|
+
...armB
|
|
1408
|
+
}, null, 2)}\n`);
|
|
1409
|
+
onProgress(`${label}: armB exit=${armB.exitCode} failureVanished=${armB.failureVanished}`);
|
|
1410
|
+
} catch (err) {
|
|
1411
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
1412
|
+
rows[index] = {
|
|
1413
|
+
...row,
|
|
1414
|
+
fix: {
|
|
1415
|
+
attempted: true,
|
|
1416
|
+
sampledOut: false,
|
|
1417
|
+
command: generated.value.command,
|
|
1418
|
+
llmError: null,
|
|
1419
|
+
usage: generated.value.usage,
|
|
1420
|
+
armBExit: null,
|
|
1421
|
+
armBPrefixExecuted: null,
|
|
1422
|
+
armBPrefixDivergences: null,
|
|
1423
|
+
armBPrefixDivergencePct: null,
|
|
1424
|
+
failureVanished: null,
|
|
1425
|
+
armBError: message.slice(0, 500),
|
|
1426
|
+
attempts: null,
|
|
1427
|
+
flippedAtAttempt: null
|
|
1428
|
+
}
|
|
1429
|
+
};
|
|
1430
|
+
onProgress(`${label}: arm B error — ${message.slice(0, 200)}`);
|
|
1431
|
+
}
|
|
1432
|
+
}
|
|
1433
|
+
llm = {
|
|
1434
|
+
model: options.fixModelLabel ?? "unknown",
|
|
1435
|
+
calls,
|
|
1436
|
+
failures,
|
|
1437
|
+
promptTokens,
|
|
1438
|
+
completionTokens,
|
|
1439
|
+
callsWithoutUsage
|
|
1440
|
+
};
|
|
1441
|
+
}
|
|
1442
|
+
const executedRows = rows.filter((r) => r.status === "ok");
|
|
1443
|
+
const replayed = executedRows.filter((r) => r.replayed);
|
|
1444
|
+
const sumOver = (pick) => executedRows.reduce((total, row) => total + (pick(row) ?? 0), 0);
|
|
1445
|
+
const executedSteps = sumOver((r) => r.prefixExecuted);
|
|
1446
|
+
const divergentSteps = sumOver((r) => r.prefixDivergences);
|
|
1447
|
+
const prefixFidelity = {
|
|
1448
|
+
executedSteps,
|
|
1449
|
+
divergentSteps,
|
|
1450
|
+
returncodeMismatches: sumOver((r) => r.prefixReturncodeMismatches),
|
|
1451
|
+
unknownExpectations: sumOver((r) => r.prefixUnknownExpectations),
|
|
1452
|
+
divergencePct: executedSteps > 0 ? Number((divergentSteps / executedSteps * 100).toFixed(1)) : null,
|
|
1453
|
+
tolerancePct: 10,
|
|
1454
|
+
casesWithinTolerance: executedRows.filter((r) => r.prefixDivergencePct !== null && r.prefixDivergencePct <= 10).length,
|
|
1455
|
+
casesExecuted: executedRows.length
|
|
1456
|
+
};
|
|
1457
|
+
const signatureStrict = executedRows.filter((r) => r.replayed && r.armASignatureMatch);
|
|
1458
|
+
const armBExecuted = rows.filter((r) => r.fix?.attempted && r.fix.failureVanished !== null);
|
|
1459
|
+
const flipped = armBExecuted.filter((r) => r.fix.failureVanished === true);
|
|
1460
|
+
const armBNonzeroRc = armBExecuted.filter((r) => r.recordedReturncodeAtK !== null && r.recordedReturncodeAtK !== 0);
|
|
1461
|
+
const flippedNonzeroRc = armBNonzeroRc.filter((r) => r.fix.failureVanished === true);
|
|
1462
|
+
const attempt1Executed = rows.filter((r) => r.fix?.attempts?.find((a) => a.attempt === 1)?.executed === true);
|
|
1463
|
+
const flippedAt1 = rows.filter((r) => r.fix?.flippedAtAttempt === 1);
|
|
1464
|
+
const flipsByAttempt = {};
|
|
1465
|
+
for (const row of rows) {
|
|
1466
|
+
const at = row.fix?.flippedAtAttempt;
|
|
1467
|
+
if (typeof at === "number") flipsByAttempt[String(at)] = (flipsByAttempt[String(at)] ?? 0) + 1;
|
|
1468
|
+
}
|
|
1469
|
+
const excludedByReason = {};
|
|
1470
|
+
for (const excluded of enumeration.excluded) excludedByReason[excluded.reason] = (excludedByReason[excluded.reason] ?? 0) + 1;
|
|
1471
|
+
const submitGoldsByCorpus = {};
|
|
1472
|
+
const submitEntry = (corpus) => submitGoldsByCorpus[corpus] ??= {
|
|
1473
|
+
submitOnlyCases: 0,
|
|
1474
|
+
goldsSkippedWithinReplayable: 0
|
|
1475
|
+
};
|
|
1476
|
+
for (const excluded of enumeration.excluded) if (excluded.reason === "gold-only-submit-step") submitEntry(excluded.corpus).submitOnlyCases += 1;
|
|
1477
|
+
for (const replayCase of enumeration.replayable) if (replayCase.submitGoldsSkipped > 0) submitEntry(replayCase.corpus).goldsSkippedWithinReplayable += replayCase.submitGoldsSkipped;
|
|
1478
|
+
const report = {
|
|
1479
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
1480
|
+
corpora: options.corpora.map((c) => ({
|
|
1481
|
+
name: c.name,
|
|
1482
|
+
labelsPath: c.labelsPath,
|
|
1483
|
+
preparedDir: c.preparedDir
|
|
1484
|
+
})),
|
|
1485
|
+
totals: {
|
|
1486
|
+
labelEntries: enumeration.labelEntryCount,
|
|
1487
|
+
replayable: enumeration.replayable.length,
|
|
1488
|
+
executed: selected.length,
|
|
1489
|
+
excludedByReason,
|
|
1490
|
+
submitGoldsByCorpus
|
|
1491
|
+
},
|
|
1492
|
+
headline: {
|
|
1493
|
+
replayabilityRate: {
|
|
1494
|
+
numerator: replayed.length,
|
|
1495
|
+
denominator: selected.length,
|
|
1496
|
+
value: rate(replayed.length, selected.length)
|
|
1497
|
+
},
|
|
1498
|
+
signatureStrictRate: {
|
|
1499
|
+
numerator: signatureStrict.length,
|
|
1500
|
+
denominator: selected.length,
|
|
1501
|
+
value: rate(signatureStrict.length, selected.length)
|
|
1502
|
+
},
|
|
1503
|
+
prefixFidelity,
|
|
1504
|
+
fixFlipRate: options.fix !== "none" ? {
|
|
1505
|
+
numerator: flipped.length,
|
|
1506
|
+
denominator: armBExecuted.length,
|
|
1507
|
+
value: rate(flipped.length, armBExecuted.length)
|
|
1508
|
+
} : null,
|
|
1509
|
+
fixFlipRateNonzeroRc: options.fix !== "none" ? {
|
|
1510
|
+
numerator: flippedNonzeroRc.length,
|
|
1511
|
+
denominator: armBNonzeroRc.length,
|
|
1512
|
+
value: rate(flippedNonzeroRc.length, armBNonzeroRc.length)
|
|
1513
|
+
} : null,
|
|
1514
|
+
fixFlipAttempt1: options.fix === "loop" ? {
|
|
1515
|
+
numerator: flippedAt1.length,
|
|
1516
|
+
denominator: attempt1Executed.length,
|
|
1517
|
+
value: rate(flippedAt1.length, attempt1Executed.length)
|
|
1518
|
+
} : null,
|
|
1519
|
+
flipsByAttempt: options.fix === "loop" ? flipsByAttempt : null
|
|
1520
|
+
},
|
|
1521
|
+
llm,
|
|
1522
|
+
excluded: enumeration.excluded,
|
|
1523
|
+
pullFailures,
|
|
1524
|
+
cases: rows
|
|
1525
|
+
};
|
|
1526
|
+
writeFileSync(join(options.out, "batch-report.json"), `${JSON.stringify(report, null, 2)}\n`);
|
|
1527
|
+
writeFileSync(join(options.out, "batch-report.md"), renderBatchReport(report));
|
|
1528
|
+
return report;
|
|
1529
|
+
}
|
|
1530
|
+
function pct(value) {
|
|
1531
|
+
return value === null ? "—" : `${(value * 100).toFixed(1)}%`;
|
|
1532
|
+
}
|
|
1533
|
+
function renderBatchReport(report) {
|
|
1534
|
+
const lines = [];
|
|
1535
|
+
const { headline, totals } = report;
|
|
1536
|
+
lines.push("# Replay-verify batch report");
|
|
1537
|
+
lines.push("");
|
|
1538
|
+
lines.push(`Generated ${report.generatedAt}.`);
|
|
1539
|
+
lines.push("");
|
|
1540
|
+
lines.push("## Headline");
|
|
1541
|
+
lines.push("");
|
|
1542
|
+
const fidelity = headline.prefixFidelity;
|
|
1543
|
+
lines.push(`- **Replayability rate: ${pct(headline.replayabilityRate.value)}** (${headline.replayabilityRate.numerator}/${headline.replayabilityRate.denominator} replayable cases where the prefix replayed within ${fidelity.tolerancePct}% divergence AND arm A reproduced the recorded returncode at the gold step k).`);
|
|
1544
|
+
lines.push(`- Signature-strict rate: ${pct(headline.signatureStrictRate.value)} (${headline.signatureStrictRate.numerator}/${headline.signatureStrictRate.denominator}; additionally requires the recorded error substring in arm A output).`);
|
|
1545
|
+
lines.push(`- **Prefix divergence: ${fidelity.divergencePct === null ? "—" : `${fidelity.divergencePct}%`}** (${fidelity.divergentSteps}/${fidelity.executedSteps} executed prefix steps did not confirm the recording: ${fidelity.returncodeMismatches} returncode mismatches, ${fidelity.unknownExpectations} with no recorded returncode to check). ${fidelity.casesWithinTolerance}/${fidelity.casesExecuted} executed cases are within the ${fidelity.tolerancePct}% tolerance.`);
|
|
1546
|
+
if (headline.fixFlipRate) lines.push(`- **Fix-flip rate: ${pct(headline.fixFlipRate.value)}** (${headline.fixFlipRate.numerator}/${headline.fixFlipRate.denominator} arm-B-executed cases where the generated fix made the failure vanish).`);
|
|
1547
|
+
if (headline.fixFlipRateNonzeroRc) lines.push(`- Fix-flip rate on recorded-rc≠0 cases: ${pct(headline.fixFlipRateNonzeroRc.value)} (${headline.fixFlipRateNonzeroRc.numerator}/${headline.fixFlipRateNonzeroRc.denominator}; real recorded failures — a gold step recorded with rc 0 flips vacuously).`);
|
|
1548
|
+
if (headline.fixFlipAttempt1) lines.push(`- Fix-flip@1: ${pct(headline.fixFlipAttempt1.value)} (${headline.fixFlipAttempt1.numerator}/${headline.fixFlipAttempt1.denominator} cases whose attempt 1 executed — the one-shot-comparable number).`);
|
|
1549
|
+
if (headline.flipsByAttempt && Object.keys(headline.flipsByAttempt).length > 0) {
|
|
1550
|
+
const parts = Object.entries(headline.flipsByAttempt).sort((a, b) => Number(a[0]) - Number(b[0])).map(([attempt, count]) => `attempt ${attempt}: ${count}`);
|
|
1551
|
+
lines.push(`- Flips by attempt: ${parts.join(", ")}.`);
|
|
1552
|
+
}
|
|
1553
|
+
lines.push("");
|
|
1554
|
+
lines.push("## Enumeration");
|
|
1555
|
+
lines.push("");
|
|
1556
|
+
lines.push(`${totals.labelEntries} label entries across ${report.corpora.length} corpora → ${totals.replayable} replayable (SWE-style docker image + ≥1 gold incorrect step), ${totals.executed} executed.`);
|
|
1557
|
+
lines.push("");
|
|
1558
|
+
lines.push("| exclusion reason | count |");
|
|
1559
|
+
lines.push("| --- | --- |");
|
|
1560
|
+
for (const [reason, count] of Object.entries(totals.excludedByReason).sort((a, b) => b[1] - a[1])) lines.push(`| ${reason} | ${count} |`);
|
|
1561
|
+
const submitEntries = Object.entries(totals.submitGoldsByCorpus);
|
|
1562
|
+
if (submitEntries.length > 0) {
|
|
1563
|
+
lines.push("");
|
|
1564
|
+
lines.push("Submit-command golds are never counterfactual targets (a gold on the submit step marks a bad submit decision, not a failed command):");
|
|
1565
|
+
lines.push("");
|
|
1566
|
+
lines.push("| corpus | cases excluded (all golds = submit) | golds skipped within replayable cases |");
|
|
1567
|
+
lines.push("| --- | --- | --- |");
|
|
1568
|
+
for (const [corpus, stats] of submitEntries.sort((a, b) => a[0].localeCompare(b[0]))) lines.push(`| ${corpus} | ${stats.submitOnlyCases} | ${stats.goldsSkippedWithinReplayable} |`);
|
|
1569
|
+
}
|
|
1570
|
+
if (report.pullFailures.length > 0) {
|
|
1571
|
+
lines.push("");
|
|
1572
|
+
lines.push("## Image pull/build failures");
|
|
1573
|
+
lines.push("");
|
|
1574
|
+
lines.push("| corpus | trajectory | image | error |");
|
|
1575
|
+
lines.push("| --- | --- | --- | --- |");
|
|
1576
|
+
for (const failure of report.pullFailures) lines.push(`| ${failure.corpus} | ${failure.trajId} | \`${failure.image}\` | ${failure.error.replaceAll("|", "\\|")} |`);
|
|
1577
|
+
}
|
|
1578
|
+
lines.push("");
|
|
1579
|
+
lines.push("## Per-case results");
|
|
1580
|
+
lines.push("");
|
|
1581
|
+
lines.push("| corpus | trajectory | k | rc@k | prefix | confirmed | rc mismatch | unknown rc | div | div% | armA exit | rc match | sig match | replayed | fix | armB exit | armB div% | vanished | wall s |");
|
|
1582
|
+
lines.push(`| ${Array(19).fill("---").join(" | ")} |`);
|
|
1583
|
+
for (const row of report.cases) {
|
|
1584
|
+
const fix = row.fix;
|
|
1585
|
+
const fixCell = !fix ? "—" : fix.sampledOut ? "sampled-out" : fix.llmError ? "llm-failed" : fix.armBError ? "armB-error" : fix.attempts ? fix.flippedAtAttempt !== null ? `flip@${fix.flippedAtAttempt}` : `exhausted(${fix.attempts.length})` : "generated";
|
|
1586
|
+
lines.push([
|
|
1587
|
+
row.corpus,
|
|
1588
|
+
row.trajId.length > 48 ? `${row.trajId.slice(0, 45)}…` : row.trajId,
|
|
1589
|
+
row.k,
|
|
1590
|
+
row.recordedReturncodeAtK ?? "null",
|
|
1591
|
+
row.prefixExecuted ?? "—",
|
|
1592
|
+
row.prefixConfirmed ?? "—",
|
|
1593
|
+
row.prefixReturncodeMismatches ?? "—",
|
|
1594
|
+
row.prefixUnknownExpectations ?? "—",
|
|
1595
|
+
row.prefixDivergences ?? "—",
|
|
1596
|
+
row.prefixDivergencePct ?? "—",
|
|
1597
|
+
row.status === "ok" ? row.armAExit : row.status,
|
|
1598
|
+
row.armAReturncodeMatch ? "yes" : "no",
|
|
1599
|
+
row.armASignatureMatch ? "yes" : "no",
|
|
1600
|
+
row.replayed ? "**yes**" : "no",
|
|
1601
|
+
fixCell,
|
|
1602
|
+
fix?.armBExit ?? "—",
|
|
1603
|
+
fix?.armBPrefixDivergencePct ?? "—",
|
|
1604
|
+
fix?.failureVanished === null || fix === null ? "—" : fix.failureVanished ? "**yes**" : "no",
|
|
1605
|
+
(row.wallMs / 1e3).toFixed(1)
|
|
1606
|
+
].join(" | "));
|
|
1607
|
+
}
|
|
1608
|
+
if (report.llm) {
|
|
1609
|
+
lines.push("");
|
|
1610
|
+
lines.push("## LLM fix generation");
|
|
1611
|
+
lines.push("");
|
|
1612
|
+
lines.push(`Model ${report.llm.model}: ${report.llm.calls} calls (${report.llm.failures} failed), ${report.llm.promptTokens} prompt + ${report.llm.completionTokens} completion tokens.`);
|
|
1613
|
+
}
|
|
1614
|
+
lines.push("");
|
|
1615
|
+
return lines.join("\n");
|
|
1616
|
+
}
|
|
1617
|
+
//#endregion
|
|
1618
|
+
//#region src/trajectory-replay/wire.ts
|
|
1619
|
+
/**
|
|
1620
|
+
* Finding-to-replay wire: the entry point that turns a cited incorrect-steps
|
|
1621
|
+
* finding into an executed replay proof.
|
|
1622
|
+
*
|
|
1623
|
+
* An analyst finding over a labeled trajectory corpus names its trajectory and
|
|
1624
|
+
* carries a subject of the form
|
|
1625
|
+
* `incorrect-steps-<first>-<last>-<escaped|unescaped>-consequence-<step>`.
|
|
1626
|
+
* The wire maps that to a replay invocation: the executed step is the
|
|
1627
|
+
* finding's first incorrect step (which may differ from the gold label), the
|
|
1628
|
+
* image and cwd come from the trajectory's raw config, and — optionally — an
|
|
1629
|
+
* arm-B corrected command comes from the counterfactual fix generator.
|
|
1630
|
+
*/
|
|
1631
|
+
const SUBJECT_PATTERN = /^incorrect-steps-(\d+)-(\d+)-(escaped|unescaped)-consequence-(\d+)$/;
|
|
1632
|
+
/** Null when the subject is not an incorrect-steps finding subject. */
|
|
1633
|
+
function parseIncorrectStepsSubject(subject) {
|
|
1634
|
+
const match = SUBJECT_PATTERN.exec(subject);
|
|
1635
|
+
if (!match) return null;
|
|
1636
|
+
return {
|
|
1637
|
+
firstStep: Number(match[1]),
|
|
1638
|
+
lastStep: Number(match[2]),
|
|
1639
|
+
escapeStatus: match[3],
|
|
1640
|
+
consequenceStep: Number(match[4])
|
|
1641
|
+
};
|
|
1642
|
+
}
|
|
1643
|
+
/**
|
|
1644
|
+
* Maps a finding onto replay resources, searching the given corpora for the
|
|
1645
|
+
* trajectory. Throws with the precise reason when the finding cannot be
|
|
1646
|
+
* replayed (malformed subject, unknown trajectory, non-replayable case, step
|
|
1647
|
+
* out of range) — the caller surfaces that reason instead of a proof.
|
|
1648
|
+
*/
|
|
1649
|
+
function resolveFindingInvocation(finding, corpora) {
|
|
1650
|
+
const subject = parseIncorrectStepsSubject(finding.subject);
|
|
1651
|
+
if (!subject) throw new Error(`trajectory-replay: subject '${finding.subject}' is not incorrect-steps-<first>-<last>-<escaped|unescaped>-consequence-<step>`);
|
|
1652
|
+
const failures = [];
|
|
1653
|
+
for (const corpus of corpora) {
|
|
1654
|
+
const resolution = resolveCaseResources(corpus, finding.trajId);
|
|
1655
|
+
if (resolution.resolved) {
|
|
1656
|
+
const at = subject.firstStep;
|
|
1657
|
+
if (!resolution.resources.steps.find((s) => s.step_id === at)) throw new Error(`trajectory-replay: finding step ${at} is outside ${finding.trajId} (${resolution.resources.steps.length} steps)`);
|
|
1658
|
+
return {
|
|
1659
|
+
resources: resolution.resources,
|
|
1660
|
+
subject,
|
|
1661
|
+
at
|
|
1662
|
+
};
|
|
1663
|
+
}
|
|
1664
|
+
failures.push(`${corpus.name}: ${resolution.reason}`);
|
|
1665
|
+
}
|
|
1666
|
+
throw new Error(`trajectory-replay: trajectory ${finding.trajId} is not replayable in any corpus — ${failures.join("; ")}`);
|
|
1667
|
+
}
|
|
1668
|
+
/**
|
|
1669
|
+
* Finding in, executed proof out. The image comes from the corpus resources;
|
|
1670
|
+
* the backend factory receives it as-is, so a factory backed by infrastructure
|
|
1671
|
+
* that needs a derived image must run an `ImagePreparer` first.
|
|
1672
|
+
*/
|
|
1673
|
+
async function replayVerifyFinding(finding, options) {
|
|
1674
|
+
if (options.fixCaller && options.fixCommand !== void 0) throw new Error("trajectory-replay: pass fixCaller or fixCommand, not both");
|
|
1675
|
+
const invocation = resolveFindingInvocation(finding, options.corpora);
|
|
1676
|
+
let fixCommand = options.fixCommand ?? null;
|
|
1677
|
+
if (options.fixCaller) {
|
|
1678
|
+
const generated = await generateFixCommand(options.fixCaller, {
|
|
1679
|
+
taskStatement: invocation.resources.taskStatement,
|
|
1680
|
+
steps: invocation.resources.steps,
|
|
1681
|
+
k: invocation.at
|
|
1682
|
+
});
|
|
1683
|
+
if (!generated.succeeded) throw new Error(`trajectory-replay: fix generation failed — ${generated.error}`);
|
|
1684
|
+
fixCommand = generated.value.command;
|
|
1685
|
+
}
|
|
1686
|
+
const verdict = await replayVerify({
|
|
1687
|
+
stepsPath: invocation.resources.stepsPath,
|
|
1688
|
+
image: invocation.resources.image,
|
|
1689
|
+
at: invocation.at,
|
|
1690
|
+
fixCommand: fixCommand ?? void 0,
|
|
1691
|
+
cwd: invocation.resources.cwd,
|
|
1692
|
+
out: options.out,
|
|
1693
|
+
caseId: invocation.resources.trajId,
|
|
1694
|
+
stepTimeoutMs: options.stepTimeoutMs ?? invocation.resources.recordedStepTimeoutMs ?? void 0,
|
|
1695
|
+
prefixLimit: options.prefixLimit,
|
|
1696
|
+
backend: options.backendFactory(invocation.resources.image),
|
|
1697
|
+
onProgress: options.onProgress
|
|
1698
|
+
});
|
|
1699
|
+
return {
|
|
1700
|
+
invocation,
|
|
1701
|
+
fixCommand,
|
|
1702
|
+
verdict
|
|
1703
|
+
};
|
|
1704
|
+
}
|
|
1705
|
+
//#endregion
|
|
1706
|
+
//#region src/trajectory-replay/findings.ts
|
|
1707
|
+
/**
|
|
1708
|
+
* Proof-carrying findings: execute analyst findings as replays.
|
|
1709
|
+
*
|
|
1710
|
+
* An analyst finding is a cited claim ("step 12 is where the run went
|
|
1711
|
+
* wrong"). This module turns each finding into an executed verdict by
|
|
1712
|
+
* replaying the trajectory prefix and running the accused step (arm A),
|
|
1713
|
+
* optionally followed by a corrected step (arm B):
|
|
1714
|
+
*
|
|
1715
|
+
* `reproduced` arm A re-produced the recorded failure signature
|
|
1716
|
+
* (returncode + stable output substring).
|
|
1717
|
+
* `fix-flipped` arm A reproduced AND arm B's corrected command made the
|
|
1718
|
+
* failure vanish — the strongest per-finding proof.
|
|
1719
|
+
* `divergent` arm A executed but the recorded failure did NOT
|
|
1720
|
+
* reproduce; evidence against the finding (or against
|
|
1721
|
+
* replay fidelity — the receipt carries prefix
|
|
1722
|
+
* divergences so the reader can tell which).
|
|
1723
|
+
* `not-replayable` the finding could not be executed at all; the receipt
|
|
1724
|
+
* carries the precise reason (no step subject, unknown
|
|
1725
|
+
* trajectory, no recorded image, submit step, …).
|
|
1726
|
+
*
|
|
1727
|
+
* Verification is execution, not generation: no LLM is involved unless the
|
|
1728
|
+
* caller supplies a corrected command for arm B. A caller whose execution
|
|
1729
|
+
* environment must be reachable before proofs run passes `preflight`, which
|
|
1730
|
+
* fails loud instead of letting verification skip silently.
|
|
1731
|
+
*
|
|
1732
|
+
* Findings are matched by the shape the analyst product emits
|
|
1733
|
+
* (`AnalystFinding`): `subject` (`incorrect-step-<n>` or the wire's
|
|
1734
|
+
* `incorrect-steps-<f>-<l>-<escaped|unescaped>-consequence-<c>`),
|
|
1735
|
+
* `metadata.block_first_step`, and `trace://<traj>/…` evidence refs.
|
|
1736
|
+
*/
|
|
1737
|
+
const SINGLE_STEP_SUBJECT = /^incorrect-step-(\d+)$/;
|
|
1738
|
+
/**
|
|
1739
|
+
* 1-based step the finding accuses, or null when the finding names none.
|
|
1740
|
+
* `metadata.block_first_step` wins over the subject: the analyst records the
|
|
1741
|
+
* block's first incorrect step there even when the subject names a later
|
|
1742
|
+
* step of the same block.
|
|
1743
|
+
*/
|
|
1744
|
+
function findingReplayStep(finding) {
|
|
1745
|
+
const fromMetadata = finding.metadata?.block_first_step;
|
|
1746
|
+
if (typeof fromMetadata === "number" && Number.isInteger(fromMetadata) && fromMetadata >= 1) return fromMetadata;
|
|
1747
|
+
const subject = finding.subject ?? "";
|
|
1748
|
+
const wire = parseIncorrectStepsSubject(subject);
|
|
1749
|
+
if (wire) return wire.firstStep;
|
|
1750
|
+
const single = SINGLE_STEP_SUBJECT.exec(subject);
|
|
1751
|
+
if (single) return Number(single[1]);
|
|
1752
|
+
return null;
|
|
1753
|
+
}
|
|
1754
|
+
const TRACE_EVIDENCE_URI = /^trace:\/\/([^/]+)\//;
|
|
1755
|
+
/** Trajectory id from the finding's `trace://<traj>/…` evidence refs, or null. */
|
|
1756
|
+
function findingTrajectoryId(finding) {
|
|
1757
|
+
for (const ref of finding.evidence_refs ?? []) {
|
|
1758
|
+
if (typeof ref.uri !== "string") continue;
|
|
1759
|
+
const match = TRACE_EVIDENCE_URI.exec(ref.uri);
|
|
1760
|
+
if (match) return match[1];
|
|
1761
|
+
}
|
|
1762
|
+
return null;
|
|
1763
|
+
}
|
|
1764
|
+
function checkStep(steps, at) {
|
|
1765
|
+
const step = steps.find((s) => s.step_id === at);
|
|
1766
|
+
if (!step) return {
|
|
1767
|
+
ok: false,
|
|
1768
|
+
reason: `step ${at} is outside the trajectory (${steps.length} steps)`
|
|
1769
|
+
};
|
|
1770
|
+
if (isSubmitAction(step.action)) return {
|
|
1771
|
+
ok: false,
|
|
1772
|
+
reason: `step ${at} is the submit action — a submit decision has no executable failure to replay`
|
|
1773
|
+
};
|
|
1774
|
+
const recordedReturncode = parseRecordedReturncode(step.observation);
|
|
1775
|
+
if (recordedReturncode === null) return {
|
|
1776
|
+
ok: false,
|
|
1777
|
+
reason: `step ${at} recorded no returncode — there is no executable failure signature to reproduce`
|
|
1778
|
+
};
|
|
1779
|
+
return {
|
|
1780
|
+
ok: true,
|
|
1781
|
+
recordedReturncode
|
|
1782
|
+
};
|
|
1783
|
+
}
|
|
1784
|
+
/**
|
|
1785
|
+
* Decides whether one finding can be executed against the source, and with
|
|
1786
|
+
* what invocation. Never throws for a finding-shaped problem — every dead end
|
|
1787
|
+
* becomes a `not-replayable` reason the receipt can carry verbatim.
|
|
1788
|
+
*/
|
|
1789
|
+
function resolveFindingReplayability(finding, source) {
|
|
1790
|
+
const at = findingReplayStep(finding);
|
|
1791
|
+
if (at === null) return {
|
|
1792
|
+
replayable: false,
|
|
1793
|
+
reason: `subject '${finding.subject ?? "(none)"}' names no trajectory step (expected incorrect-step-<n>, incorrect-steps-<f>-<l>-…, or metadata.block_first_step)`
|
|
1794
|
+
};
|
|
1795
|
+
if (source.kind === "direct") {
|
|
1796
|
+
const trajId = findingTrajectoryId(finding);
|
|
1797
|
+
if (trajId && source.caseId && trajId !== source.caseId) return {
|
|
1798
|
+
replayable: false,
|
|
1799
|
+
reason: `finding cites trajectory '${trajId}' but the supplied steps are case '${source.caseId}'`
|
|
1800
|
+
};
|
|
1801
|
+
let steps;
|
|
1802
|
+
try {
|
|
1803
|
+
steps = JSON.parse(readFileSync(source.stepsPath, "utf8"));
|
|
1804
|
+
} catch (err) {
|
|
1805
|
+
throw new Error(`verify-findings: cannot read steps file ${source.stepsPath} — ${err instanceof Error ? err.message : String(err)}`);
|
|
1806
|
+
}
|
|
1807
|
+
if (!Array.isArray(steps) || steps.length === 0) throw new Error(`verify-findings: ${source.stepsPath} is not a non-empty steps array`);
|
|
1808
|
+
const step = checkStep(steps, at);
|
|
1809
|
+
if (!step.ok) return {
|
|
1810
|
+
replayable: false,
|
|
1811
|
+
reason: step.reason
|
|
1812
|
+
};
|
|
1813
|
+
return {
|
|
1814
|
+
replayable: true,
|
|
1815
|
+
resolved: {
|
|
1816
|
+
caseId: source.caseId ?? trajId ?? source.stepsPath,
|
|
1817
|
+
stepsPath: source.stepsPath,
|
|
1818
|
+
image: source.image,
|
|
1819
|
+
cwd: source.cwd,
|
|
1820
|
+
at,
|
|
1821
|
+
recordedReturncode: step.recordedReturncode,
|
|
1822
|
+
recordedStepTimeoutMs: null
|
|
1823
|
+
}
|
|
1824
|
+
};
|
|
1825
|
+
}
|
|
1826
|
+
const trajId = findingTrajectoryId(finding);
|
|
1827
|
+
if (!trajId) return {
|
|
1828
|
+
replayable: false,
|
|
1829
|
+
reason: "finding carries no trace://<trajectory>/ evidence ref naming its trajectory"
|
|
1830
|
+
};
|
|
1831
|
+
const failures = [];
|
|
1832
|
+
for (const corpus of source.corpora) {
|
|
1833
|
+
const resolution = resolveCaseResources(corpus, trajId);
|
|
1834
|
+
if (!resolution.resolved) {
|
|
1835
|
+
failures.push(`${corpus.name}: ${resolution.reason}${resolution.detail ? ` (${resolution.detail})` : ""}`);
|
|
1836
|
+
continue;
|
|
1837
|
+
}
|
|
1838
|
+
const step = checkStep(resolution.resources.steps, at);
|
|
1839
|
+
if (!step.ok) return {
|
|
1840
|
+
replayable: false,
|
|
1841
|
+
reason: step.reason
|
|
1842
|
+
};
|
|
1843
|
+
return {
|
|
1844
|
+
replayable: true,
|
|
1845
|
+
resolved: {
|
|
1846
|
+
caseId: trajId,
|
|
1847
|
+
stepsPath: resolution.resources.stepsPath,
|
|
1848
|
+
image: resolution.resources.image,
|
|
1849
|
+
cwd: resolution.resources.cwd,
|
|
1850
|
+
at,
|
|
1851
|
+
recordedReturncode: step.recordedReturncode,
|
|
1852
|
+
recordedStepTimeoutMs: resolution.resources.recordedStepTimeoutMs
|
|
1853
|
+
}
|
|
1854
|
+
};
|
|
1855
|
+
}
|
|
1856
|
+
return {
|
|
1857
|
+
replayable: false,
|
|
1858
|
+
reason: `trajectory ${trajId} is not replayable in any corpus — ${failures.join("; ")}`
|
|
1859
|
+
};
|
|
1860
|
+
}
|
|
1861
|
+
/**
|
|
1862
|
+
* Arm A reproduced on a prefix the recording confirmed → the fix flipping it
|
|
1863
|
+
* beats plain reproduction; anything else diverged. A proof standing on a
|
|
1864
|
+
* prefix outside the divergence tolerance is divergent no matter what arm A
|
|
1865
|
+
* did: the state it ran against is not the recorded state.
|
|
1866
|
+
*/
|
|
1867
|
+
function classifyVerdict(verdict) {
|
|
1868
|
+
if (!verdict.prefixWithinTolerance) return "divergent";
|
|
1869
|
+
if (!verdict.armA.failureSignatureMatch) return "divergent";
|
|
1870
|
+
if (verdict.armB?.failureVanished) return "fix-flipped";
|
|
1871
|
+
return "reproduced";
|
|
1872
|
+
}
|
|
1873
|
+
function receiptExecution(verdict) {
|
|
1874
|
+
return {
|
|
1875
|
+
image: verdict.image,
|
|
1876
|
+
cwd: verdict.cwd,
|
|
1877
|
+
recordedReturncode: verdict.recordedReturncode,
|
|
1878
|
+
signature: verdict.signature,
|
|
1879
|
+
signatureBasis: verdict.signatureBasis,
|
|
1880
|
+
armA: {
|
|
1881
|
+
command: verdict.armA.command,
|
|
1882
|
+
exitCode: verdict.armA.exitCode,
|
|
1883
|
+
wallMs: verdict.armA.wallMs,
|
|
1884
|
+
failureSignatureMatch: verdict.armA.failureSignatureMatch
|
|
1885
|
+
},
|
|
1886
|
+
armB: verdict.armB ? {
|
|
1887
|
+
command: verdict.armB.command,
|
|
1888
|
+
exitCode: verdict.armB.exitCode,
|
|
1889
|
+
wallMs: verdict.armB.wallMs,
|
|
1890
|
+
failureVanished: verdict.armB.failureVanished
|
|
1891
|
+
} : null,
|
|
1892
|
+
prefixExecuted: verdict.prefixExecuted,
|
|
1893
|
+
prefixDivergences: verdict.prefixDivergences.length,
|
|
1894
|
+
prefixDivergencePct: verdict.prefixDivergencePct,
|
|
1895
|
+
prefixReturncodeMismatches: verdict.prefixReturncodeMismatches,
|
|
1896
|
+
prefixUnknownExpectations: verdict.prefixUnknownExpectations,
|
|
1897
|
+
prefixWithinTolerance: verdict.prefixWithinTolerance,
|
|
1898
|
+
totalMs: verdict.timings.totalMs
|
|
1899
|
+
};
|
|
1900
|
+
}
|
|
1901
|
+
function writeReceipt(receiptDir, finding, verification, execution) {
|
|
1902
|
+
const receipt = {
|
|
1903
|
+
schema_version: "1.0.0",
|
|
1904
|
+
produced_at: (/* @__PURE__ */ new Date()).toISOString(),
|
|
1905
|
+
finding_id: verification.finding_id,
|
|
1906
|
+
analyst_id: finding.analyst_id ?? null,
|
|
1907
|
+
subject: verification.subject,
|
|
1908
|
+
claim: typeof finding.claim === "string" ? finding.claim.slice(0, 600) : null,
|
|
1909
|
+
trajectory_id: verification.trajectory_id,
|
|
1910
|
+
step: verification.step,
|
|
1911
|
+
verified: verification.verified,
|
|
1912
|
+
reason: verification.reason,
|
|
1913
|
+
execution,
|
|
1914
|
+
verdict_path: verification.verdict_path,
|
|
1915
|
+
deduplicated_with: verification.deduplicated_with
|
|
1916
|
+
};
|
|
1917
|
+
writeFileSync(join(receiptDir, "receipt.json"), `${JSON.stringify(receipt, null, 2)}\n`);
|
|
1918
|
+
}
|
|
1919
|
+
function receiptDirName(index, finding) {
|
|
1920
|
+
const id = typeof finding.finding_id === "string" && finding.finding_id.length > 0 ? finding.finding_id.replace(/[^A-Za-z0-9_-]/g, "_") : "finding";
|
|
1921
|
+
return `${String(index + 1).padStart(3, "0")}-${id}`;
|
|
1922
|
+
}
|
|
1923
|
+
/**
|
|
1924
|
+
* Verifies every finding against the source: resolves replayability, executes
|
|
1925
|
+
* one proof per distinct (case, step, fix) — findings accusing the same step
|
|
1926
|
+
* share the executed proof — and writes a receipt directory per finding plus a
|
|
1927
|
+
* run-level verifications.json.
|
|
1928
|
+
*/
|
|
1929
|
+
async function verifyFindings(findings, options) {
|
|
1930
|
+
if (findings.length === 0) throw new Error("verify-findings: no findings to verify");
|
|
1931
|
+
mkdirSync(options.out, { recursive: true });
|
|
1932
|
+
const resolutions = findings.map((finding) => resolveFindingReplayability(finding, options.source));
|
|
1933
|
+
if (resolutions.some((resolution) => resolution.replayable) && options.preflight) await options.preflight();
|
|
1934
|
+
const preparer = options.source.kind === "corpus" ? options.source.preparer === void 0 ? dockerImagePreparer() : options.source.preparer : null;
|
|
1935
|
+
const preparedImages = /* @__PURE__ */ new Map();
|
|
1936
|
+
const executedByKey = /* @__PURE__ */ new Map();
|
|
1937
|
+
const verifications = [];
|
|
1938
|
+
let executions = 0;
|
|
1939
|
+
for (let index = 0; index < findings.length; index++) {
|
|
1940
|
+
const finding = findings[index];
|
|
1941
|
+
const resolution = resolutions[index];
|
|
1942
|
+
const receiptDir = join(options.out, receiptDirName(index, finding));
|
|
1943
|
+
mkdirSync(receiptDir, { recursive: true });
|
|
1944
|
+
const identity = {
|
|
1945
|
+
finding_id: finding.finding_id ?? null,
|
|
1946
|
+
subject: finding.subject ?? null,
|
|
1947
|
+
trajectory_id: findingTrajectoryId(finding)
|
|
1948
|
+
};
|
|
1949
|
+
if (!resolution.replayable) {
|
|
1950
|
+
const verification = {
|
|
1951
|
+
...identity,
|
|
1952
|
+
step: findingReplayStep(finding),
|
|
1953
|
+
verified: "not-replayable",
|
|
1954
|
+
reason: resolution.reason,
|
|
1955
|
+
receipt: receiptDir,
|
|
1956
|
+
verdict_path: null,
|
|
1957
|
+
deduplicated_with: null
|
|
1958
|
+
};
|
|
1959
|
+
writeReceipt(receiptDir, finding, verification, null);
|
|
1960
|
+
verifications.push(verification);
|
|
1961
|
+
options.onProgress?.(`${identity.finding_id ?? `finding ${index + 1}`}: not-replayable — ${resolution.reason}`);
|
|
1962
|
+
continue;
|
|
1963
|
+
}
|
|
1964
|
+
const resolved = resolution.resolved;
|
|
1965
|
+
const dedupeKey = `${resolved.caseId}::${resolved.at}::${options.fixCommand ?? ""}`;
|
|
1966
|
+
const prior = executedByKey.get(dedupeKey);
|
|
1967
|
+
if (prior) {
|
|
1968
|
+
const verification = {
|
|
1969
|
+
...identity,
|
|
1970
|
+
step: resolved.at,
|
|
1971
|
+
verified: prior.verified,
|
|
1972
|
+
reason: prior.reason,
|
|
1973
|
+
receipt: receiptDir,
|
|
1974
|
+
verdict_path: prior.verdict_path,
|
|
1975
|
+
deduplicated_with: prior.receipt
|
|
1976
|
+
};
|
|
1977
|
+
const priorVerdict = prior.verdict_path ? JSON.parse(readFileSync(prior.verdict_path, "utf8")) : null;
|
|
1978
|
+
writeReceipt(receiptDir, finding, verification, priorVerdict ? receiptExecution(priorVerdict) : null);
|
|
1979
|
+
verifications.push(verification);
|
|
1980
|
+
options.onProgress?.(`${identity.finding_id ?? `finding ${index + 1}`}: ${prior.verified} (shares proof with ${prior.finding_id ?? prior.receipt})`);
|
|
1981
|
+
continue;
|
|
1982
|
+
}
|
|
1983
|
+
let image = resolved.image;
|
|
1984
|
+
if (preparer) {
|
|
1985
|
+
const preparationKey = `${resolved.image}::${resolved.cwd}`;
|
|
1986
|
+
let preparation = preparedImages.get(preparationKey);
|
|
1987
|
+
if (!preparation) {
|
|
1988
|
+
const ensured = await preparer.ensure(resolved.image, resolved.cwd);
|
|
1989
|
+
preparation = ensured.succeeded ? {
|
|
1990
|
+
succeeded: true,
|
|
1991
|
+
image: ensured.value.derivedImage
|
|
1992
|
+
} : {
|
|
1993
|
+
succeeded: false,
|
|
1994
|
+
error: ensured.error
|
|
1995
|
+
};
|
|
1996
|
+
preparedImages.set(preparationKey, preparation);
|
|
1997
|
+
}
|
|
1998
|
+
if (!preparation.succeeded) {
|
|
1999
|
+
const verification = {
|
|
2000
|
+
...identity,
|
|
2001
|
+
step: resolved.at,
|
|
2002
|
+
verified: "not-replayable",
|
|
2003
|
+
reason: `replay image could not be prepared — ${preparation.error}`,
|
|
2004
|
+
receipt: receiptDir,
|
|
2005
|
+
verdict_path: null,
|
|
2006
|
+
deduplicated_with: null
|
|
2007
|
+
};
|
|
2008
|
+
writeReceipt(receiptDir, finding, verification, null);
|
|
2009
|
+
verifications.push(verification);
|
|
2010
|
+
options.onProgress?.(`${identity.finding_id ?? `finding ${index + 1}`}: not-replayable — image preparation failed`);
|
|
2011
|
+
continue;
|
|
2012
|
+
}
|
|
2013
|
+
image = preparation.image;
|
|
2014
|
+
}
|
|
2015
|
+
options.onProgress?.(`${identity.finding_id ?? `finding ${index + 1}`}: executing arm A at step ${resolved.at} of ${resolved.caseId} on ${image}`);
|
|
2016
|
+
const verdict = await replayVerify({
|
|
2017
|
+
stepsPath: resolved.stepsPath,
|
|
2018
|
+
image,
|
|
2019
|
+
at: resolved.at,
|
|
2020
|
+
fixCommand: options.fixCommand,
|
|
2021
|
+
cwd: resolved.cwd,
|
|
2022
|
+
out: receiptDir,
|
|
2023
|
+
caseId: resolved.caseId,
|
|
2024
|
+
stepTimeoutMs: options.stepTimeoutMs ?? resolved.recordedStepTimeoutMs ?? void 0,
|
|
2025
|
+
prefixLimit: options.prefixLimit,
|
|
2026
|
+
backend: options.backendFactory(image),
|
|
2027
|
+
onProgress: options.onProgress
|
|
2028
|
+
});
|
|
2029
|
+
executions += 1;
|
|
2030
|
+
const verification = {
|
|
2031
|
+
...identity,
|
|
2032
|
+
step: resolved.at,
|
|
2033
|
+
verified: classifyVerdict(verdict),
|
|
2034
|
+
reason: null,
|
|
2035
|
+
receipt: receiptDir,
|
|
2036
|
+
verdict_path: join(receiptDir, "replay-verdict.json"),
|
|
2037
|
+
deduplicated_with: null
|
|
2038
|
+
};
|
|
2039
|
+
executedByKey.set(dedupeKey, verification);
|
|
2040
|
+
writeReceipt(receiptDir, finding, verification, receiptExecution(verdict));
|
|
2041
|
+
verifications.push(verification);
|
|
2042
|
+
options.onProgress?.(`${identity.finding_id ?? `finding ${index + 1}`}: ${verification.verified}`);
|
|
2043
|
+
}
|
|
2044
|
+
const counts = {
|
|
2045
|
+
reproduced: 0,
|
|
2046
|
+
"fix-flipped": 0,
|
|
2047
|
+
divergent: 0,
|
|
2048
|
+
"not-replayable": 0
|
|
2049
|
+
};
|
|
2050
|
+
for (const verification of verifications) counts[verification.verified] += 1;
|
|
2051
|
+
const run = {
|
|
2052
|
+
out: options.out,
|
|
2053
|
+
verifications,
|
|
2054
|
+
counts,
|
|
2055
|
+
executions
|
|
2056
|
+
};
|
|
2057
|
+
writeFileSync(join(options.out, "verifications.json"), `${JSON.stringify({
|
|
2058
|
+
schema_version: "1.0.0",
|
|
2059
|
+
...run
|
|
2060
|
+
}, null, 2)}\n`);
|
|
2061
|
+
return run;
|
|
2062
|
+
}
|
|
2063
|
+
function verdictCell(verification) {
|
|
2064
|
+
switch (verification.verified) {
|
|
2065
|
+
case "reproduced": return "**VERIFIED** — reproduced";
|
|
2066
|
+
case "fix-flipped": return "**VERIFIED** — fix-flipped";
|
|
2067
|
+
case "divergent": return "DIVERGENT — recorded failure did not reproduce";
|
|
2068
|
+
case "not-replayable": return `UNVERIFIABLE — ${verification.reason ?? "no reason recorded"}`;
|
|
2069
|
+
}
|
|
2070
|
+
}
|
|
2071
|
+
/** Markdown section an analysis report appends when finding verification ran. */
|
|
2072
|
+
function renderVerifiedFindingsSection(run) {
|
|
2073
|
+
const lines = ["## Verified findings (executed replay)", ""];
|
|
2074
|
+
const total = run.verifications.length;
|
|
2075
|
+
lines.push(`${total} finding(s) → ${run.counts.reproduced} reproduced, ${run.counts["fix-flipped"]} fix-flipped, ${run.counts.divergent} divergent, ${run.counts["not-replayable"]} not replayable (${run.executions} execution(s); findings accusing the same step share one proof).`);
|
|
2076
|
+
lines.push("");
|
|
2077
|
+
lines.push("| Finding | Subject | Step | Verdict | Receipt |");
|
|
2078
|
+
lines.push("|---|---|---:|---|---|");
|
|
2079
|
+
for (const verification of run.verifications) {
|
|
2080
|
+
const shared = verification.deduplicated_with ? " (shared proof)" : "";
|
|
2081
|
+
lines.push(`| \`${verification.finding_id ?? "—"}\` | \`${verification.subject ?? "—"}\` | ${verification.step ?? "—"} | ${verdictCell(verification)} | \`${verification.receipt}\`${shared} |`);
|
|
2082
|
+
}
|
|
2083
|
+
lines.push("");
|
|
2084
|
+
lines.push("VERIFIED = the accused step was re-executed after replaying the trajectory prefix, and the recorded failure signature reproduced (fix-flipped: a corrected command additionally made it vanish). Each receipt directory carries receipt.json and, when executed, replay-verdict.json + report.md with real stdout/stderr.");
|
|
2085
|
+
lines.push("");
|
|
2086
|
+
return lines.join("\n");
|
|
2087
|
+
}
|
|
2088
|
+
/**
|
|
2089
|
+
* Accepts the two shapes findings travel in: a bare JSON array of analyst
|
|
2090
|
+
* findings, or an object with a `findings` array (e.g. an extracted
|
|
2091
|
+
* `observations[n]` from a result.json).
|
|
2092
|
+
*/
|
|
2093
|
+
function readFindingsFile(path) {
|
|
2094
|
+
const parsed = JSON.parse(readFileSync(path, "utf8"));
|
|
2095
|
+
const array = Array.isArray(parsed) ? parsed : parsed && typeof parsed === "object" && Array.isArray(parsed.findings) ? parsed.findings : null;
|
|
2096
|
+
if (!array) throw new Error(`${path} is neither a findings array nor an object with a findings array`);
|
|
2097
|
+
for (const entry of array) if (!entry || typeof entry !== "object") throw new Error(`${path}: every finding must be an object, got ${JSON.stringify(entry)}`);
|
|
2098
|
+
return array;
|
|
2099
|
+
}
|
|
2100
|
+
//#endregion
|
|
2101
|
+
export { PREFIX_DIVERGENCE_TOLERANCE_PCT, SUBMIT_ACTION_SIGNATURE, SandboxCounterfactualRunner, buildFixPrompt, buildRetryFixPrompt, classifyPrefixStep, classifyVerdict, clipText, countScriptCommands, deriveFailureSignature, derivedImageTag, dockerImagePreparer, enumerateReplayableCases, extractFixCommand, findingReplayStep, findingTrajectoryId, generateFixCommand, goldIncorrectSteps, ingestRecordedTrajectory, isSubmitAction, parseCorpusFlag, parseIncorrectStepsSubject, parseObservationOutput, parseRecordedReturncode, readFindingsFile, readLabelEntries, renderBatchReport, renderVerifiedFindingsSection, replayVerify, replayVerifyFinding, resolveCaseResources, resolveFindingInvocation, resolveFindingReplayability, runFixLoop, runReplayBatch, seededSample, summarizePrefixReplay, verifyFindings, wrapActionForExec };
|
|
2102
|
+
|
|
2103
|
+
//# sourceMappingURL=index.js.map
|