@tangle-network/agent-eval 0.144.5 → 0.144.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -0
- package/dist/{benchmark-Fmo42QVE.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +471 -89
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CA_NFOmy.js → benchmark-command-BCafwNrf.js} +664 -605
- package/dist/benchmark-command-BCafwNrf.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CWbsj0t4.js → benchmarks-CDSolHq7.js} +4 -4
- package/dist/{benchmarks-CWbsj0t4.js.map → benchmarks-CDSolHq7.js.map} +1 -1
- package/dist/campaign/index.d.ts +8 -6
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-ClpnD7Ug.js → campaign-Tdy3h62h.js} +17 -301
- package/dist/campaign-Tdy3h62h.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-DAb7MWtL.d.ts → client-DjXROWpx.d.ts} +4 -4
- package/dist/{client-DAb7MWtL.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-VvpHRu78.d.ts → completion-verifier-foUCLif_.d.ts} +6 -6
- package/dist/{completion-verifier-VvpHRu78.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +11 -10
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
- package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
- package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
- package/dist/{default-registry-BwbZ9N9v.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-RLNNoeEP.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-RLNNoeEP.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-ByZiqpVk.d.ts → engine-nB64f48I.d.ts} +21 -34
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
- package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
- package/dist/errors-D-LKuDhb.js.map +1 -1
- package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-BQ7W90C4.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-BQ7W90C4.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts → feedback-trajectory-Rh280oXo.d.ts} +4 -4
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-CsuAo2-J.d.ts → index-C5HOo4ZF2.d.ts} +5 -5
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
- package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
- package/dist/{index-CNOCxBLh.d.ts → index-CvXXlyz7.d.ts} +3 -3
- package/dist/{index-CNOCxBLh.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-DGIzNtRv.d.ts → index-CwDrUMe0.d.ts} +3 -3
- package/dist/{index-DGIzNtRv.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-CpxZSlB7.d.ts → index-Sh2I0DRc.d.ts} +15 -649
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +252 -451
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +271 -751
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-C6h6F_4L.d.ts} +4 -4
- package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-DRXobPEs.d.ts → integrity-BuqEKu-x.d.ts} +3 -3
- package/dist/{integrity-DRXobPEs.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/{llm-client-D3EoChAU.js → llm-client-DzvMUsS_.js} +310 -10
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +3 -3
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-BZRWdq_t.d.ts → release-report-CI8uisI1.d.ts} +4 -4
- package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-B7S7Pdbw.d.ts → replay-DFf-teiC.d.ts} +8 -7
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +2 -2
- package/dist/{researcher-BSiCoM1s.d.ts → researcher-BoaxeCzP.d.ts} +6 -6
- package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-Cf1PtEOz.d.ts} +34 -4
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +19 -9
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
- package/dist/{run-evidence-j5Ynww6L.d.ts → run-evidence-BDFFai9R.d.ts} +3 -3
- package/dist/{run-evidence-j5Ynww6L.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
- package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DKRtp2sY.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DKRtp2sY.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-Df00sdwz.js → server-iu0ede49.js} +2 -2
- package/dist/{server-Df00sdwz.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-DUvvudWR.d.ts → skill-usage-CJlWEUFt.d.ts} +11 -11
- package/dist/{skill-usage-DUvvudWR.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-CGz9ywhM.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +11 -297
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
- package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-B4u_CiFd.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-DuUS_i7W.d.ts} +4 -115
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-BjMFz88h.d.ts → types-D216SgwM.d.ts} +228 -9
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-CZt1PBIk.d.ts → types-DF_Udrp-.d.ts} +54 -5
- package/dist/{types-CZt1PBIk.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-Dcoaqcsc.d.ts → types-DYuNHo9R.d.ts} +5 -5
- package/dist/{types-Dcoaqcsc.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/building-doctrine.md +15 -0
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +21 -4
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-Fmo42QVE.d.ts.map +0 -1
- package/dist/benchmark-command-CA_NFOmy.js.map +0 -1
- package/dist/campaign-ClpnD7Ug.js.map +0 -1
- package/dist/default-registry-BwbZ9N9v.d.ts.map +0 -1
- package/dist/index-CpxZSlB7.d.ts.map +0 -1
- package/dist/index-CsuAo2-J.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-D3EoChAU.js.map +0 -1
- package/dist/replay-B7S7Pdbw.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-DFgkEY4p.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-C4FX42dy.js.map +0 -1
- package/dist/skillopt-optimization-method-CGz9ywhM.d.ts.map +0 -1
- package/dist/summary-report-BOM6dfP7.d.ts.map +0 -1
- package/dist/tool-groups-ByZiqpVk.d.ts.map +0 -1
- package/dist/types-BjMFz88h.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
|
@@ -0,0 +1,781 @@
|
|
|
1
|
+
import { s as TraceStore } from "../store-CT9YIIve.js";
|
|
2
|
+
import { i as TraceEmitter } from "../emitter-DGQGoLyj.js";
|
|
3
|
+
import { i as CounterfactualRunner, t as CounterfactualContext } from "../counterfactual-CxmxAONP.js";
|
|
4
|
+
import { a as parseObservationOutput, i as isSubmitAction, n as SUBMIT_ACTION_SIGNATURE, o as parseRecordedReturncode, r as deriveFailureSignature, t as RecordedTrajectoryStep } from "../steps-BArUxhna.js";
|
|
5
|
+
//#region src/trajectory-replay/corpus.d.ts
|
|
6
|
+
interface CorpusSpec {
|
|
7
|
+
readonly name: string;
|
|
8
|
+
readonly labelsPath: string;
|
|
9
|
+
readonly preparedDir: string;
|
|
10
|
+
}
|
|
11
|
+
/** Parses `name=<labelsPath>::<preparedDir>` (paths may contain `=`, not `::`). */
|
|
12
|
+
declare function parseCorpusFlag(value: string): CorpusSpec;
|
|
13
|
+
type ReplayExclusionReason = 'no-swe-raw-trajectory' | 'ambiguous-swe-raw-trajectory' | 'unreadable-raw-trajectory' | 'no-docker-image' | 'no-gold-incorrect-step' | 'gold-only-submit-step' | 'missing-steps-json' | 'gold-step-outside-steps' | 'cwd-underivable';
|
|
14
|
+
interface ExcludedCase {
|
|
15
|
+
readonly corpus: string;
|
|
16
|
+
readonly trajId: string;
|
|
17
|
+
readonly reason: ReplayExclusionReason;
|
|
18
|
+
readonly detail?: string;
|
|
19
|
+
}
|
|
20
|
+
type CwdSource = 'run-config' | 'docker-config' | 'pwd-observation';
|
|
21
|
+
/** Everything needed to replay one trajectory. */
|
|
22
|
+
interface CaseResources {
|
|
23
|
+
readonly corpus: string;
|
|
24
|
+
readonly trajId: string;
|
|
25
|
+
readonly stepsPath: string;
|
|
26
|
+
readonly steps: readonly RecordedTrajectoryStep[];
|
|
27
|
+
readonly taskStatement: string | null;
|
|
28
|
+
readonly image: string;
|
|
29
|
+
readonly cwd: string;
|
|
30
|
+
readonly cwdSource: CwdSource;
|
|
31
|
+
/** Per-step timeout the recorded run used, when the raw config carries one. */
|
|
32
|
+
readonly recordedStepTimeoutMs: number | null;
|
|
33
|
+
}
|
|
34
|
+
interface ReplayableCase extends CaseResources {
|
|
35
|
+
/** 1-based gold incorrect step ids, ascending. */
|
|
36
|
+
readonly goldIncorrectSteps: readonly number[];
|
|
37
|
+
/** k — the first gold incorrect step that is a real mid-trajectory action;
|
|
38
|
+
* submit-command golds are skipped (see SUBMIT_ACTION_SIGNATURE). */
|
|
39
|
+
readonly k: number;
|
|
40
|
+
/** Gold steps before k skipped because their action is the submit command. */
|
|
41
|
+
readonly submitGoldsSkipped: number;
|
|
42
|
+
/** Recorded returncode at k; null when the observation carries none. */
|
|
43
|
+
readonly recordedReturncodeAtK: number | null;
|
|
44
|
+
}
|
|
45
|
+
interface LabelEntry {
|
|
46
|
+
readonly traj_id: string;
|
|
47
|
+
readonly incorrect_stages?: readonly {
|
|
48
|
+
readonly incorrect_step_ids?: readonly number[];
|
|
49
|
+
}[];
|
|
50
|
+
}
|
|
51
|
+
declare function readLabelEntries(labelsPath: string): LabelEntry[];
|
|
52
|
+
declare function goldIncorrectSteps(entry: LabelEntry): number[];
|
|
53
|
+
type ResourceResolution = {
|
|
54
|
+
readonly resolved: true;
|
|
55
|
+
readonly resources: CaseResources;
|
|
56
|
+
} | {
|
|
57
|
+
readonly resolved: false;
|
|
58
|
+
readonly reason: ReplayExclusionReason;
|
|
59
|
+
readonly detail?: string;
|
|
60
|
+
};
|
|
61
|
+
/**
|
|
62
|
+
* Resolves the replay resources for one trajectory, independent of gold
|
|
63
|
+
* labels — the finding wire uses this with a finding-supplied step instead.
|
|
64
|
+
*/
|
|
65
|
+
declare function resolveCaseResources(corpus: CorpusSpec, trajId: string): ResourceResolution;
|
|
66
|
+
interface EnumerationResult {
|
|
67
|
+
readonly replayable: ReplayableCase[];
|
|
68
|
+
readonly excluded: ExcludedCase[];
|
|
69
|
+
readonly labelEntryCount: number;
|
|
70
|
+
}
|
|
71
|
+
/**
|
|
72
|
+
* Replayable = a raw trajectory with a recorded image AND at least one gold
|
|
73
|
+
* incorrect step that is a real mid-trajectory action. Gold steps whose action
|
|
74
|
+
* is the submit command are skipped when choosing k (counted per case); a case
|
|
75
|
+
* whose golds are ALL submit steps is excluded as gold-only-submit-step.
|
|
76
|
+
* Exclusion reasons are reported in resolution order:
|
|
77
|
+
* raw trajectory → image → gold labels → steps.json → k range → cwd.
|
|
78
|
+
*/
|
|
79
|
+
declare function enumerateReplayableCases(corpora: readonly CorpusSpec[]): EnumerationResult;
|
|
80
|
+
//#endregion
|
|
81
|
+
//#region src/trajectory-replay/exec.d.ts
|
|
82
|
+
/**
|
|
83
|
+
* The execution boundary replay runs across.
|
|
84
|
+
*
|
|
85
|
+
* A replay needs one thing from its environment: a session that runs a shell
|
|
86
|
+
* command inside the trajectory's own image and reports the exit code and
|
|
87
|
+
* output. That is the whole contract. Concrete backends — a sandbox platform
|
|
88
|
+
* client, a docker exec, an SSH shell — live with the consumer that owns the
|
|
89
|
+
* infrastructure, so this package depends on no sandbox client.
|
|
90
|
+
*/
|
|
91
|
+
interface ReplayExecResult {
|
|
92
|
+
exitCode: number;
|
|
93
|
+
stdout: string;
|
|
94
|
+
stderr: string;
|
|
95
|
+
}
|
|
96
|
+
interface ReplayExecSession {
|
|
97
|
+
exec(command: string, timeoutMs: number): Promise<ReplayExecResult>;
|
|
98
|
+
close(): Promise<void>;
|
|
99
|
+
}
|
|
100
|
+
interface ReplayExecBackend {
|
|
101
|
+
/** One fresh execution environment per call; the caller closes it. */
|
|
102
|
+
open(): Promise<ReplayExecSession>;
|
|
103
|
+
}
|
|
104
|
+
/** Builds a backend pinned to one image. Callers that resolve images
|
|
105
|
+
* internally (batch, corpus wire, finding verification) take this instead of
|
|
106
|
+
* a backend, so every case runs against its own image. */
|
|
107
|
+
type ReplayExecBackendFactory = (image: string) => ReplayExecBackend;
|
|
108
|
+
/**
|
|
109
|
+
* mini-SWE runs every action as a fresh /bin/sh subshell from a fixed
|
|
110
|
+
* workdir. Reproduce that exactly — and stay quote-proof for arbitrary
|
|
111
|
+
* recorded actions — by piping the base64 of the action into `sh` after
|
|
112
|
+
* cd-ing to the workdir. Exit code is sh's, i.e. the action's.
|
|
113
|
+
*/
|
|
114
|
+
declare function wrapActionForExec(action: string, cwd: string): string;
|
|
115
|
+
//#endregion
|
|
116
|
+
//#region src/trajectory-replay/fix.d.ts
|
|
117
|
+
interface ChatUsage {
|
|
118
|
+
readonly promptTokens: number;
|
|
119
|
+
readonly completionTokens: number;
|
|
120
|
+
}
|
|
121
|
+
type ChatOutcome = {
|
|
122
|
+
readonly succeeded: true;
|
|
123
|
+
readonly value: {
|
|
124
|
+
content: string;
|
|
125
|
+
usage: ChatUsage | null;
|
|
126
|
+
};
|
|
127
|
+
} | {
|
|
128
|
+
readonly succeeded: false;
|
|
129
|
+
readonly error: string;
|
|
130
|
+
};
|
|
131
|
+
interface ChatCompletionCaller {
|
|
132
|
+
complete(system: string, user: string): Promise<ChatOutcome>;
|
|
133
|
+
}
|
|
134
|
+
interface FixPromptInput {
|
|
135
|
+
readonly taskStatement: string | null;
|
|
136
|
+
readonly steps: readonly RecordedTrajectoryStep[];
|
|
137
|
+
/** 1-based step_id of the incorrect step. */
|
|
138
|
+
readonly k: number;
|
|
139
|
+
/** Steps of context on each side of k (default 3). */
|
|
140
|
+
readonly contextRadius?: number;
|
|
141
|
+
}
|
|
142
|
+
/** Head+tail excerpt with an elision marker; identity below the limit. */
|
|
143
|
+
declare function clipText(text: string, limit: number): string;
|
|
144
|
+
declare function buildFixPrompt(input: FixPromptInput): {
|
|
145
|
+
system: string;
|
|
146
|
+
user: string;
|
|
147
|
+
};
|
|
148
|
+
/** One prior attempt of the fix loop, rendered into the retry prompt. */
|
|
149
|
+
interface FailedFixAttempt {
|
|
150
|
+
readonly attempt: number;
|
|
151
|
+
/** Null when the model call itself failed before producing a command. */
|
|
152
|
+
readonly command: string | null;
|
|
153
|
+
readonly exitCode: number | null;
|
|
154
|
+
readonly stdoutTail: string | null;
|
|
155
|
+
readonly stderrTail: string | null;
|
|
156
|
+
readonly llmError: string | null;
|
|
157
|
+
}
|
|
158
|
+
/**
|
|
159
|
+
* Retry prompt for fix-loop attempts ≥2: the original context plus every prior
|
|
160
|
+
* attempt with its REAL executed output, and permission to answer with a short
|
|
161
|
+
* script (the block still executes as one /bin/sh unit).
|
|
162
|
+
*/
|
|
163
|
+
declare function buildRetryFixPrompt(input: FixPromptInput, priorAttempts: readonly FailedFixAttempt[], maxScriptCommands?: number): {
|
|
164
|
+
system: string;
|
|
165
|
+
user: string;
|
|
166
|
+
};
|
|
167
|
+
/** Non-empty, non-comment lines of a fix script — the loop's script-size cap. */
|
|
168
|
+
declare function countScriptCommands(script: string): number;
|
|
169
|
+
/** Last fenced code block, else the whole trimmed content; null when empty. */
|
|
170
|
+
declare function extractFixCommand(content: string): string | null;
|
|
171
|
+
type FixGenerationOutcome = {
|
|
172
|
+
readonly succeeded: true;
|
|
173
|
+
readonly value: {
|
|
174
|
+
command: string;
|
|
175
|
+
usage: ChatUsage | null;
|
|
176
|
+
};
|
|
177
|
+
} | {
|
|
178
|
+
readonly succeeded: false;
|
|
179
|
+
readonly error: string;
|
|
180
|
+
};
|
|
181
|
+
declare function generateFixCommand(caller: ChatCompletionCaller, input: FixPromptInput): Promise<FixGenerationOutcome>;
|
|
182
|
+
//#endregion
|
|
183
|
+
//#region src/trajectory-replay/fix-loop.d.ts
|
|
184
|
+
/** Result of executing one corrected command as a full arm. */
|
|
185
|
+
interface FixArmExecution {
|
|
186
|
+
readonly exitCode: number;
|
|
187
|
+
readonly prefixExecuted: number;
|
|
188
|
+
readonly prefixDivergences: number;
|
|
189
|
+
/** Divergent prefix steps over executed ones, in percent. */
|
|
190
|
+
readonly prefixDivergencePct: number;
|
|
191
|
+
readonly failureVanished: boolean;
|
|
192
|
+
readonly stdout: string;
|
|
193
|
+
readonly stderr: string;
|
|
194
|
+
}
|
|
195
|
+
/** Runs one corrected command as a full arm: fresh sandbox + prefix replay. */
|
|
196
|
+
type FixArmExecutor = (command: string, attempt: number) => Promise<FixArmExecution>;
|
|
197
|
+
interface FixLoopOptions {
|
|
198
|
+
/** Total LLM attempts per case (≥1); 1 degenerates to the one-shot path. */
|
|
199
|
+
readonly maxAttempts: number;
|
|
200
|
+
/** Command-line cap on retry scripts (default 5). */
|
|
201
|
+
readonly maxScriptCommands?: number;
|
|
202
|
+
/** Chars kept per stdout/stderr tail in records and retry prompts. */
|
|
203
|
+
readonly outputTailChars?: number;
|
|
204
|
+
readonly onProgress?: (message: string) => void;
|
|
205
|
+
}
|
|
206
|
+
interface FixLoopAttemptRecord {
|
|
207
|
+
readonly attempt: number;
|
|
208
|
+
readonly command: string | null;
|
|
209
|
+
readonly llmError: string | null;
|
|
210
|
+
readonly usage: {
|
|
211
|
+
readonly promptTokens: number;
|
|
212
|
+
readonly completionTokens: number;
|
|
213
|
+
} | null;
|
|
214
|
+
readonly executed: boolean;
|
|
215
|
+
readonly exitCode: number | null;
|
|
216
|
+
readonly prefixExecuted: number | null;
|
|
217
|
+
readonly prefixDivergences: number | null;
|
|
218
|
+
readonly prefixDivergencePct: number | null;
|
|
219
|
+
readonly failureVanished: boolean | null;
|
|
220
|
+
readonly stdoutTail: string | null;
|
|
221
|
+
readonly stderrTail: string | null;
|
|
222
|
+
readonly armBError: string | null;
|
|
223
|
+
readonly wallMs: number;
|
|
224
|
+
}
|
|
225
|
+
interface FixLoopResult {
|
|
226
|
+
readonly flipped: boolean;
|
|
227
|
+
readonly flippedAtAttempt: number | null;
|
|
228
|
+
readonly attempts: readonly FixLoopAttemptRecord[];
|
|
229
|
+
/** True when a sandbox error ended the loop before the attempt budget. */
|
|
230
|
+
readonly aborted: boolean;
|
|
231
|
+
readonly llmCalls: number;
|
|
232
|
+
/** Calls that produced no runnable fix: transport errors, empty replies,
|
|
233
|
+
* and retry scripts over the command cap. */
|
|
234
|
+
readonly llmFailures: number;
|
|
235
|
+
readonly promptTokens: number;
|
|
236
|
+
readonly completionTokens: number;
|
|
237
|
+
/** Successful calls whose provider reported no usage. Their tokens are absent
|
|
238
|
+
* from the two totals above, so a nonzero count makes those totals a lower
|
|
239
|
+
* bound rather than a measurement. */
|
|
240
|
+
readonly callsWithoutUsage: number;
|
|
241
|
+
}
|
|
242
|
+
declare function runFixLoop(caller: ChatCompletionCaller, input: FixPromptInput, executor: FixArmExecutor, options: FixLoopOptions): Promise<FixLoopResult>;
|
|
243
|
+
//#endregion
|
|
244
|
+
//#region src/trajectory-replay/image-preparer.d.ts
|
|
245
|
+
/**
|
|
246
|
+
* Replay-ready image derivation.
|
|
247
|
+
*
|
|
248
|
+
* A recorded trajectory names the image it ran in, but the image is not always
|
|
249
|
+
* runnable as-is: sandbox platforms pin customer commands to a non-root
|
|
250
|
+
* identity, so a root-owned working tree must be chowned before the replay can
|
|
251
|
+
* write to it. `ImagePreparer` is that step, injectable so a consumer whose
|
|
252
|
+
* images are already replay-ready supplies its own no-op or none at all.
|
|
253
|
+
*/
|
|
254
|
+
type ImagePreparation = {
|
|
255
|
+
readonly succeeded: true;
|
|
256
|
+
readonly value: {
|
|
257
|
+
derivedImage: string;
|
|
258
|
+
pulled: boolean;
|
|
259
|
+
built: boolean;
|
|
260
|
+
};
|
|
261
|
+
} | {
|
|
262
|
+
readonly succeeded: false;
|
|
263
|
+
readonly error: string;
|
|
264
|
+
};
|
|
265
|
+
interface ImagePreparer {
|
|
266
|
+
ensure(image: string, cwd: string): Promise<ImagePreparation>;
|
|
267
|
+
}
|
|
268
|
+
interface DockerImagePreparerOptions {
|
|
269
|
+
readonly pullTimeoutMs?: number;
|
|
270
|
+
readonly buildTimeoutMs?: number;
|
|
271
|
+
}
|
|
272
|
+
declare function derivedImageTag(image: string, cwd: string): string;
|
|
273
|
+
/**
|
|
274
|
+
* Pulls the base image when absent and builds `FROM <base>; RUN chown -R
|
|
275
|
+
* 1000:1000 <cwd>` tagged by content hash, so repeated batches reuse both
|
|
276
|
+
* the pull and the build. cwd `/` skips the chown (never chown -R /) and
|
|
277
|
+
* replays on the base image directly.
|
|
278
|
+
*/
|
|
279
|
+
declare function dockerImagePreparer(options?: DockerImagePreparerOptions): ImagePreparer;
|
|
280
|
+
//#endregion
|
|
281
|
+
//#region src/trajectory-replay/batch.d.ts
|
|
282
|
+
interface ReplayBatchOptions {
|
|
283
|
+
readonly corpora: readonly CorpusSpec[];
|
|
284
|
+
readonly out: string;
|
|
285
|
+
/** 'generate' = one LLM call per arm-A-reproduced case, then arm B.
|
|
286
|
+
* 'loop' = iterative: failed arms feed their real output into up to
|
|
287
|
+
* `fixAttempts` prompts, each executed in its own fresh session. */
|
|
288
|
+
readonly fix: 'none' | 'generate' | 'loop';
|
|
289
|
+
/** Attempt budget per case in loop mode (default 3). */
|
|
290
|
+
readonly fixAttempts?: number;
|
|
291
|
+
readonly fixCaller?: ChatCompletionCaller;
|
|
292
|
+
readonly fixModelLabel?: string;
|
|
293
|
+
/** Cap on LLM fix calls; eligible cases beyond it are seeded-sampled out. */
|
|
294
|
+
readonly maxFixCases?: number;
|
|
295
|
+
readonly seed?: number;
|
|
296
|
+
/** Overrides the per-case recorded step timeout. */
|
|
297
|
+
readonly stepTimeoutMs?: number;
|
|
298
|
+
readonly prefixLimit?: number;
|
|
299
|
+
/** Run only cases whose trajId contains this substring (smoke knob). */
|
|
300
|
+
readonly caseFilter?: string;
|
|
301
|
+
/** Run only the first N replayable cases (smoke knob). */
|
|
302
|
+
readonly caseLimit?: number;
|
|
303
|
+
/** Derives the replay-ready image per case. Defaults to the docker preparer. */
|
|
304
|
+
readonly preparer?: ImagePreparer;
|
|
305
|
+
/** Builds the exec backend for a case's derived image. */
|
|
306
|
+
readonly backendFactory: ReplayExecBackendFactory;
|
|
307
|
+
readonly onProgress?: (message: string) => void;
|
|
308
|
+
}
|
|
309
|
+
interface ReplayBatchFixResult {
|
|
310
|
+
/** false when the case was eligible but seeded-sampled out of the cap. */
|
|
311
|
+
readonly attempted: boolean;
|
|
312
|
+
readonly sampledOut: boolean;
|
|
313
|
+
readonly command: string | null;
|
|
314
|
+
readonly llmError: string | null;
|
|
315
|
+
/** Loop mode: token totals summed across every attempt (null when the
|
|
316
|
+
* provider reported no usage). */
|
|
317
|
+
readonly usage: ChatUsage | null;
|
|
318
|
+
readonly armBExit: number | null;
|
|
319
|
+
readonly armBPrefixExecuted: number | null;
|
|
320
|
+
readonly armBPrefixDivergences: number | null;
|
|
321
|
+
readonly armBPrefixDivergencePct: number | null;
|
|
322
|
+
readonly failureVanished: boolean | null;
|
|
323
|
+
readonly armBError: string | null;
|
|
324
|
+
/** Loop mode only: the full per-attempt trail. Null in generate mode. */
|
|
325
|
+
readonly attempts: readonly FixLoopAttemptRecord[] | null;
|
|
326
|
+
/** 1-based attempt that flipped the failure; null when none did.
|
|
327
|
+
* Generate mode: 1 when the single attempt flipped. */
|
|
328
|
+
readonly flippedAtAttempt: number | null;
|
|
329
|
+
}
|
|
330
|
+
interface ReplayBatchCaseRow {
|
|
331
|
+
readonly corpus: string;
|
|
332
|
+
readonly trajId: string;
|
|
333
|
+
readonly image: string;
|
|
334
|
+
readonly derivedImage: string | null;
|
|
335
|
+
readonly cwd: string;
|
|
336
|
+
readonly cwdSource: string;
|
|
337
|
+
readonly k: number;
|
|
338
|
+
readonly stepCount: number;
|
|
339
|
+
readonly goldIncorrectSteps: readonly number[];
|
|
340
|
+
/** Gold steps before k skipped because their action is the submit command. */
|
|
341
|
+
readonly submitGoldsSkipped: number;
|
|
342
|
+
readonly recordedReturncodeAtK: number | null;
|
|
343
|
+
readonly signature: string | null;
|
|
344
|
+
readonly status: 'ok' | 'image-unavailable' | 'replay-error';
|
|
345
|
+
readonly error: string | null;
|
|
346
|
+
readonly imagePulled: boolean;
|
|
347
|
+
readonly imageBuilt: boolean;
|
|
348
|
+
readonly prefixExecuted: number | null;
|
|
349
|
+
readonly prefixDivergences: number | null;
|
|
350
|
+
readonly prefixDivergencePct: number | null;
|
|
351
|
+
/** Prefix steps whose recorded returncode equalled the replayed exit. */
|
|
352
|
+
readonly prefixConfirmed: number | null;
|
|
353
|
+
readonly prefixReturncodeMismatches: number | null;
|
|
354
|
+
/** Prefix steps the recording carries no returncode for: unverifiable, and
|
|
355
|
+
* counted as divergences because agreement was never established. */
|
|
356
|
+
readonly prefixUnknownExpectations: number | null;
|
|
357
|
+
readonly armAExit: number | null;
|
|
358
|
+
readonly armAReturncodeMatch: boolean;
|
|
359
|
+
readonly armASignatureMatch: boolean;
|
|
360
|
+
/** Headline predicate: prefix divergence within tolerance AND arm A
|
|
361
|
+
* reproduced the recorded returncode at k. */
|
|
362
|
+
readonly replayed: boolean;
|
|
363
|
+
readonly fix: ReplayBatchFixResult | null;
|
|
364
|
+
readonly wallMs: number;
|
|
365
|
+
}
|
|
366
|
+
interface ReplayBatchReport {
|
|
367
|
+
readonly generatedAt: string;
|
|
368
|
+
readonly corpora: readonly {
|
|
369
|
+
name: string;
|
|
370
|
+
labelsPath: string;
|
|
371
|
+
preparedDir: string;
|
|
372
|
+
}[];
|
|
373
|
+
readonly totals: {
|
|
374
|
+
readonly labelEntries: number;
|
|
375
|
+
readonly replayable: number;
|
|
376
|
+
readonly executed: number;
|
|
377
|
+
readonly excludedByReason: Record<string, number>;
|
|
378
|
+
/** Per-corpus submit-gold accounting: cases dropped because every gold is
|
|
379
|
+
* the submit command, and golds skipped inside still-replayable cases. */
|
|
380
|
+
readonly submitGoldsByCorpus: Record<string, {
|
|
381
|
+
submitOnlyCases: number;
|
|
382
|
+
goldsSkippedWithinReplayable: number;
|
|
383
|
+
}>;
|
|
384
|
+
};
|
|
385
|
+
readonly headline: {
|
|
386
|
+
readonly replayabilityRate: {
|
|
387
|
+
numerator: number;
|
|
388
|
+
denominator: number;
|
|
389
|
+
value: number | null;
|
|
390
|
+
};
|
|
391
|
+
readonly signatureStrictRate: {
|
|
392
|
+
numerator: number;
|
|
393
|
+
denominator: number;
|
|
394
|
+
value: number | null;
|
|
395
|
+
};
|
|
396
|
+
/** Corpus-level replay fidelity over every executed prefix step. A corpus
|
|
397
|
+
* whose recordings cannot adjudicate the replay lands here as
|
|
398
|
+
* `unknownExpectations`, not as a clean replay. */
|
|
399
|
+
readonly prefixFidelity: {
|
|
400
|
+
readonly executedSteps: number;
|
|
401
|
+
readonly divergentSteps: number;
|
|
402
|
+
readonly returncodeMismatches: number;
|
|
403
|
+
readonly unknownExpectations: number;
|
|
404
|
+
/** Divergent over executed steps; null when no prefix step ran. */
|
|
405
|
+
readonly divergencePct: number | null;
|
|
406
|
+
readonly tolerancePct: number;
|
|
407
|
+
readonly casesWithinTolerance: number;
|
|
408
|
+
readonly casesExecuted: number;
|
|
409
|
+
};
|
|
410
|
+
readonly fixFlipRate: {
|
|
411
|
+
numerator: number;
|
|
412
|
+
denominator: number;
|
|
413
|
+
value: number | null;
|
|
414
|
+
} | null;
|
|
415
|
+
/** Fix-flip restricted to cases whose recorded returncode at k is nonzero —
|
|
416
|
+
* real recorded failures, where "the failure vanished" is not vacuous. */
|
|
417
|
+
readonly fixFlipRateNonzeroRc: {
|
|
418
|
+
numerator: number;
|
|
419
|
+
denominator: number;
|
|
420
|
+
value: number | null;
|
|
421
|
+
} | null;
|
|
422
|
+
/** Loop mode only: flips at attempt 1 over cases whose attempt 1 executed —
|
|
423
|
+
* the number directly comparable to the one-shot fixFlipRate. */
|
|
424
|
+
readonly fixFlipAttempt1: {
|
|
425
|
+
numerator: number;
|
|
426
|
+
denominator: number;
|
|
427
|
+
value: number | null;
|
|
428
|
+
} | null;
|
|
429
|
+
/** Loop mode only: flip count keyed by the attempt number that flipped. */
|
|
430
|
+
readonly flipsByAttempt: Record<string, number> | null;
|
|
431
|
+
};
|
|
432
|
+
readonly llm: {
|
|
433
|
+
readonly model: string;
|
|
434
|
+
readonly calls: number;
|
|
435
|
+
readonly failures: number;
|
|
436
|
+
readonly promptTokens: number;
|
|
437
|
+
readonly completionTokens: number;
|
|
438
|
+
/** Successful calls whose provider reported no usage. Their tokens are
|
|
439
|
+
* absent from the two totals above, so a nonzero count here means the
|
|
440
|
+
* totals are a lower bound, not a measurement. */
|
|
441
|
+
readonly callsWithoutUsage: number;
|
|
442
|
+
} | null;
|
|
443
|
+
readonly excluded: readonly {
|
|
444
|
+
corpus: string;
|
|
445
|
+
trajId: string;
|
|
446
|
+
reason: string;
|
|
447
|
+
detail?: string;
|
|
448
|
+
}[];
|
|
449
|
+
readonly pullFailures: readonly {
|
|
450
|
+
corpus: string;
|
|
451
|
+
trajId: string;
|
|
452
|
+
image: string;
|
|
453
|
+
error: string;
|
|
454
|
+
}[];
|
|
455
|
+
readonly cases: readonly ReplayBatchCaseRow[];
|
|
456
|
+
}
|
|
457
|
+
declare function seededSample<T>(items: readonly T[], size: number, seed: number): Set<T>;
|
|
458
|
+
declare function runReplayBatch(options: ReplayBatchOptions): Promise<ReplayBatchReport>;
|
|
459
|
+
declare function renderBatchReport(report: ReplayBatchReport): string;
|
|
460
|
+
//#endregion
|
|
461
|
+
//#region src/trajectory-replay/verify.d.ts
|
|
462
|
+
interface IngestedTrajectory {
|
|
463
|
+
runId: string;
|
|
464
|
+
stepCount: number;
|
|
465
|
+
}
|
|
466
|
+
/**
|
|
467
|
+
* Emit one tool span per trajectory step, in order, with a monotonic
|
|
468
|
+
* injected clock so `buildTrajectory` ordering is deterministic even when
|
|
469
|
+
* two spans would share a Date.now() millisecond.
|
|
470
|
+
*/
|
|
471
|
+
declare function ingestRecordedTrajectory(store: TraceStore, steps: readonly RecordedTrajectoryStep[], caseId: string): Promise<IngestedTrajectory>;
|
|
472
|
+
/** Why a replayed prefix step failed to confirm the recording. */
|
|
473
|
+
type PrefixDivergenceKind =
|
|
474
|
+
/** The recording carries a returncode and the replayed exit differs from it. */
|
|
475
|
+
'returncode-mismatch' |
|
|
476
|
+
/** The recording carries no returncode, so agreement cannot be established. */
|
|
477
|
+
'unknown-expectation';
|
|
478
|
+
interface PrefixDivergence {
|
|
479
|
+
step: number;
|
|
480
|
+
kind: PrefixDivergenceKind;
|
|
481
|
+
/** null exactly when `kind` is `unknown-expectation`. */
|
|
482
|
+
expectedReturncode: number | null;
|
|
483
|
+
actualExit: number;
|
|
484
|
+
}
|
|
485
|
+
interface PrefixReplayResult {
|
|
486
|
+
prefixExecuted: number;
|
|
487
|
+
/** Every step that did not confirm the recording, of both kinds. */
|
|
488
|
+
prefixDivergences: PrefixDivergence[];
|
|
489
|
+
/** Steps whose recorded returncode equalled the replayed exit. */
|
|
490
|
+
prefixConfirmed: number;
|
|
491
|
+
prefixReturncodeMismatches: number;
|
|
492
|
+
prefixUnknownExpectations: number;
|
|
493
|
+
/** Divergent steps over executed steps, in percent, one decimal.
|
|
494
|
+
* 0 for an empty prefix: no step ran, so none diverged. */
|
|
495
|
+
prefixDivergencePct: number;
|
|
496
|
+
/** `prefixDivergencePct` within `PREFIX_DIVERGENCE_TOLERANCE_PCT`. */
|
|
497
|
+
prefixWithinTolerance: boolean;
|
|
498
|
+
wallMs: number;
|
|
499
|
+
}
|
|
500
|
+
/** Admission tolerance: a prefix replay is faithful enough to build a verdict
|
|
501
|
+
* on when at most this percentage of its executed steps diverged. */
|
|
502
|
+
declare const PREFIX_DIVERGENCE_TOLERANCE_PCT = 10;
|
|
503
|
+
/**
|
|
504
|
+
* Compare one replayed prefix step against its recording. Returns null only
|
|
505
|
+
* when the recording positively confirms the replay.
|
|
506
|
+
*/
|
|
507
|
+
declare function classifyPrefixStep(step: number, expectedReturncode: number | null, actualExit: number): PrefixDivergence | null;
|
|
508
|
+
/** Roll per-step classifications up into the rate the admission pre-pass gates on. */
|
|
509
|
+
declare function summarizePrefixReplay(prefixExecuted: number, prefixDivergences: PrefixDivergence[], wallMs: number): PrefixReplayResult;
|
|
510
|
+
interface ArmExecutionResult {
|
|
511
|
+
command: string;
|
|
512
|
+
exitCode: number;
|
|
513
|
+
stdout: string;
|
|
514
|
+
stderr: string;
|
|
515
|
+
wallMs: number;
|
|
516
|
+
}
|
|
517
|
+
interface SandboxCounterfactualRunnerOptions {
|
|
518
|
+
cwd: string;
|
|
519
|
+
stepTimeoutMs: number;
|
|
520
|
+
/** Execute at most this many prefix steps (from the start). Fast-iteration
|
|
521
|
+
* knob; a truncated prefix weakens state fidelity and the verdict says so. */
|
|
522
|
+
prefixLimit?: number;
|
|
523
|
+
onProgress?: (message: string) => void;
|
|
524
|
+
}
|
|
525
|
+
/**
|
|
526
|
+
* Replays `ctx.prefix` in a fresh session, then executes the mutated step.
|
|
527
|
+
* Divergences are recorded and never abort the replay. Results land on
|
|
528
|
+
* `lastPrefix` / `lastArm` for the caller; spans for every exec land in the
|
|
529
|
+
* counterfactual meta-run.
|
|
530
|
+
*/
|
|
531
|
+
declare class SandboxCounterfactualRunner implements CounterfactualRunner {
|
|
532
|
+
private readonly backend;
|
|
533
|
+
private readonly options;
|
|
534
|
+
lastPrefix: PrefixReplayResult | null;
|
|
535
|
+
lastArm: ArmExecutionResult | null;
|
|
536
|
+
constructor(backend: ReplayExecBackend, options: SandboxCounterfactualRunnerOptions);
|
|
537
|
+
executeFrom(ctx: CounterfactualContext, emitter: TraceEmitter): Promise<void>;
|
|
538
|
+
}
|
|
539
|
+
interface ReplayVerifyOptions {
|
|
540
|
+
stepsPath: string;
|
|
541
|
+
image: string;
|
|
542
|
+
/** 1-based step_id of the error-critical step. */
|
|
543
|
+
at: number;
|
|
544
|
+
/** Corrected command for arm B. Omit to run arm A only. */
|
|
545
|
+
fixCommand?: string;
|
|
546
|
+
cwd: string;
|
|
547
|
+
out: string;
|
|
548
|
+
caseId?: string;
|
|
549
|
+
/** Override the auto-derived failure signature substring. */
|
|
550
|
+
signature?: string;
|
|
551
|
+
stepTimeoutMs?: number;
|
|
552
|
+
prefixLimit?: number;
|
|
553
|
+
/** Label of the driver backing the exec backend (reported, not probed). */
|
|
554
|
+
driverLabel?: string;
|
|
555
|
+
/** Execution environment for both arms; each arm opens its own session. */
|
|
556
|
+
backend: ReplayExecBackend;
|
|
557
|
+
onProgress?: (message: string) => void;
|
|
558
|
+
}
|
|
559
|
+
interface ReplayArmVerdict {
|
|
560
|
+
command: string;
|
|
561
|
+
exitCode: number;
|
|
562
|
+
wallMs: number;
|
|
563
|
+
prefix: PrefixReplayResult;
|
|
564
|
+
}
|
|
565
|
+
interface ReplayVerdict {
|
|
566
|
+
case: string;
|
|
567
|
+
image: string;
|
|
568
|
+
driver: string;
|
|
569
|
+
k: number;
|
|
570
|
+
cwd: string;
|
|
571
|
+
recordedReturncode: number | null;
|
|
572
|
+
signature: string | null;
|
|
573
|
+
signatureBasis: 'returncode+output-substring' | 'returncode-only';
|
|
574
|
+
prefixExecuted: number;
|
|
575
|
+
prefixDivergences: PrefixDivergence[];
|
|
576
|
+
/** Arm A's prefix divergence rate — the number admission gates on. */
|
|
577
|
+
prefixDivergencePct: number;
|
|
578
|
+
prefixConfirmed: number;
|
|
579
|
+
prefixReturncodeMismatches: number;
|
|
580
|
+
prefixUnknownExpectations: number;
|
|
581
|
+
prefixWithinTolerance: boolean;
|
|
582
|
+
armA: ReplayArmVerdict & {
|
|
583
|
+
failureSignatureMatch: boolean;
|
|
584
|
+
};
|
|
585
|
+
armB: (ReplayArmVerdict & {
|
|
586
|
+
failureVanished: boolean;
|
|
587
|
+
}) | null;
|
|
588
|
+
timings: {
|
|
589
|
+
armAMs: number;
|
|
590
|
+
armBMs: number | null;
|
|
591
|
+
totalMs: number;
|
|
592
|
+
};
|
|
593
|
+
runIds: {
|
|
594
|
+
original: string;
|
|
595
|
+
armA: string;
|
|
596
|
+
armB: string | null;
|
|
597
|
+
};
|
|
598
|
+
}
|
|
599
|
+
declare function replayVerify(options: ReplayVerifyOptions): Promise<ReplayVerdict>;
|
|
600
|
+
//#endregion
|
|
601
|
+
//#region src/trajectory-replay/findings.d.ts
|
|
602
|
+
interface VerifiableFinding {
|
|
603
|
+
readonly finding_id?: string;
|
|
604
|
+
readonly analyst_id?: string;
|
|
605
|
+
readonly subject?: string;
|
|
606
|
+
readonly area?: string;
|
|
607
|
+
readonly claim?: string;
|
|
608
|
+
readonly evidence_refs?: readonly {
|
|
609
|
+
readonly kind?: string;
|
|
610
|
+
readonly uri?: string;
|
|
611
|
+
readonly excerpt?: string;
|
|
612
|
+
}[];
|
|
613
|
+
readonly metadata?: Readonly<Record<string, unknown>>;
|
|
614
|
+
}
|
|
615
|
+
/**
|
|
616
|
+
* 1-based step the finding accuses, or null when the finding names none.
|
|
617
|
+
* `metadata.block_first_step` wins over the subject: the analyst records the
|
|
618
|
+
* block's first incorrect step there even when the subject names a later
|
|
619
|
+
* step of the same block.
|
|
620
|
+
*/
|
|
621
|
+
declare function findingReplayStep(finding: VerifiableFinding): number | null;
|
|
622
|
+
/** Trajectory id from the finding's `trace://<traj>/…` evidence refs, or null. */
|
|
623
|
+
declare function findingTrajectoryId(finding: VerifiableFinding): string | null;
|
|
624
|
+
/**
|
|
625
|
+
* Where the executable trajectory lives.
|
|
626
|
+
* `direct` — one trajectory's steps.json plus a replay-ready image; the
|
|
627
|
+
* caller owns image preparation.
|
|
628
|
+
* `corpus` — labeled trajectory corpora; each finding's trajectory is
|
|
629
|
+
* resolved by its `trace://` evidence and the image is derived through
|
|
630
|
+
* `preparer` (the docker uid-1000 preparer unless overridden; `null` when
|
|
631
|
+
* the images are already replay-ready).
|
|
632
|
+
*/
|
|
633
|
+
type FindingReplaySource = {
|
|
634
|
+
readonly kind: 'direct';
|
|
635
|
+
readonly stepsPath: string;
|
|
636
|
+
readonly image: string;
|
|
637
|
+
readonly cwd: string;
|
|
638
|
+
readonly caseId?: string;
|
|
639
|
+
} | {
|
|
640
|
+
readonly kind: 'corpus';
|
|
641
|
+
readonly corpora: readonly CorpusSpec[];
|
|
642
|
+
readonly preparer?: ImagePreparer | null;
|
|
643
|
+
};
|
|
644
|
+
interface ResolvedFindingReplay {
|
|
645
|
+
readonly caseId: string;
|
|
646
|
+
readonly stepsPath: string;
|
|
647
|
+
/** Raw image; corpus-mode execution derives the replay image from it. */
|
|
648
|
+
readonly image: string;
|
|
649
|
+
readonly cwd: string;
|
|
650
|
+
/** 1-based step_id arm A executes. */
|
|
651
|
+
readonly at: number;
|
|
652
|
+
readonly recordedReturncode: number;
|
|
653
|
+
readonly recordedStepTimeoutMs: number | null;
|
|
654
|
+
}
|
|
655
|
+
type FindingReplayability = {
|
|
656
|
+
readonly replayable: true;
|
|
657
|
+
readonly resolved: ResolvedFindingReplay;
|
|
658
|
+
} | {
|
|
659
|
+
readonly replayable: false;
|
|
660
|
+
readonly reason: string;
|
|
661
|
+
};
|
|
662
|
+
/**
|
|
663
|
+
* Decides whether one finding can be executed against the source, and with
|
|
664
|
+
* what invocation. Never throws for a finding-shaped problem — every dead end
|
|
665
|
+
* becomes a `not-replayable` reason the receipt can carry verbatim.
|
|
666
|
+
*/
|
|
667
|
+
declare function resolveFindingReplayability(finding: VerifiableFinding, source: FindingReplaySource): FindingReplayability;
|
|
668
|
+
type FindingVerificationStatus = 'reproduced' | 'fix-flipped' | 'not-replayable' | 'divergent';
|
|
669
|
+
interface FindingVerification {
|
|
670
|
+
readonly finding_id: string | null;
|
|
671
|
+
readonly subject: string | null;
|
|
672
|
+
readonly trajectory_id: string | null;
|
|
673
|
+
/** 1-based step the proof executed at; null when not replayable. */
|
|
674
|
+
readonly step: number | null;
|
|
675
|
+
readonly verified: FindingVerificationStatus;
|
|
676
|
+
/** Present exactly when `verified` is `not-replayable`. */
|
|
677
|
+
readonly reason: string | null;
|
|
678
|
+
/** Receipt directory: receipt.json plus, when executed, replay-verdict.json + report.md. */
|
|
679
|
+
readonly receipt: string;
|
|
680
|
+
readonly verdict_path: string | null;
|
|
681
|
+
/** Receipt dir of the executed proof this finding shares (same case, step, and fix). */
|
|
682
|
+
readonly deduplicated_with: string | null;
|
|
683
|
+
}
|
|
684
|
+
interface VerifyFindingsRun {
|
|
685
|
+
readonly out: string;
|
|
686
|
+
readonly verifications: readonly FindingVerification[];
|
|
687
|
+
readonly counts: Readonly<Record<FindingVerificationStatus, number>>;
|
|
688
|
+
/** Executions actually performed (deduplicated proofs count once). */
|
|
689
|
+
readonly executions: number;
|
|
690
|
+
}
|
|
691
|
+
interface VerifyFindingsOptions {
|
|
692
|
+
readonly source: FindingReplaySource;
|
|
693
|
+
/** Receipt root; one subdirectory per finding plus verifications.json. */
|
|
694
|
+
readonly out: string;
|
|
695
|
+
/** Corrected command for arm B on every executed finding; omit for arm A only. */
|
|
696
|
+
readonly fixCommand?: string;
|
|
697
|
+
readonly stepTimeoutMs?: number;
|
|
698
|
+
readonly prefixLimit?: number;
|
|
699
|
+
/** Builds the exec backend for a finding's replay image. */
|
|
700
|
+
readonly backendFactory: ReplayExecBackendFactory;
|
|
701
|
+
/** Runs once before the first proof, only when some finding is replayable.
|
|
702
|
+
* Throw to refuse execution against absent or degraded infrastructure. */
|
|
703
|
+
readonly preflight?: () => Promise<void>;
|
|
704
|
+
readonly onProgress?: (message: string) => void;
|
|
705
|
+
}
|
|
706
|
+
/**
|
|
707
|
+
* Arm A reproduced on a prefix the recording confirmed → the fix flipping it
|
|
708
|
+
* beats plain reproduction; anything else diverged. A proof standing on a
|
|
709
|
+
* prefix outside the divergence tolerance is divergent no matter what arm A
|
|
710
|
+
* did: the state it ran against is not the recorded state.
|
|
711
|
+
*/
|
|
712
|
+
declare function classifyVerdict(verdict: ReplayVerdict): FindingVerificationStatus;
|
|
713
|
+
/**
|
|
714
|
+
* Verifies every finding against the source: resolves replayability, executes
|
|
715
|
+
* one proof per distinct (case, step, fix) — findings accusing the same step
|
|
716
|
+
* share the executed proof — and writes a receipt directory per finding plus a
|
|
717
|
+
* run-level verifications.json.
|
|
718
|
+
*/
|
|
719
|
+
declare function verifyFindings(findings: readonly VerifiableFinding[], options: VerifyFindingsOptions): Promise<VerifyFindingsRun>;
|
|
720
|
+
/** Markdown section an analysis report appends when finding verification ran. */
|
|
721
|
+
declare function renderVerifiedFindingsSection(run: VerifyFindingsRun): string;
|
|
722
|
+
/**
|
|
723
|
+
* Accepts the two shapes findings travel in: a bare JSON array of analyst
|
|
724
|
+
* findings, or an object with a `findings` array (e.g. an extracted
|
|
725
|
+
* `observations[n]` from a result.json).
|
|
726
|
+
*/
|
|
727
|
+
declare function readFindingsFile(path: string): VerifiableFinding[];
|
|
728
|
+
//#endregion
|
|
729
|
+
//#region src/trajectory-replay/wire.d.ts
|
|
730
|
+
interface IncorrectStepsSubject {
|
|
731
|
+
readonly firstStep: number;
|
|
732
|
+
readonly lastStep: number;
|
|
733
|
+
readonly escapeStatus: 'escaped' | 'unescaped';
|
|
734
|
+
readonly consequenceStep: number;
|
|
735
|
+
}
|
|
736
|
+
/** Null when the subject is not an incorrect-steps finding subject. */
|
|
737
|
+
declare function parseIncorrectStepsSubject(subject: string): IncorrectStepsSubject | null;
|
|
738
|
+
interface AnalystReplayFinding {
|
|
739
|
+
readonly trajId: string;
|
|
740
|
+
readonly subject: string;
|
|
741
|
+
}
|
|
742
|
+
interface ResolvedReplayInvocation {
|
|
743
|
+
readonly resources: CaseResources;
|
|
744
|
+
readonly subject: IncorrectStepsSubject;
|
|
745
|
+
/** 1-based step_id arm A executes: the finding's first incorrect step. */
|
|
746
|
+
readonly at: number;
|
|
747
|
+
}
|
|
748
|
+
/**
|
|
749
|
+
* Maps a finding onto replay resources, searching the given corpora for the
|
|
750
|
+
* trajectory. Throws with the precise reason when the finding cannot be
|
|
751
|
+
* replayed (malformed subject, unknown trajectory, non-replayable case, step
|
|
752
|
+
* out of range) — the caller surfaces that reason instead of a proof.
|
|
753
|
+
*/
|
|
754
|
+
declare function resolveFindingInvocation(finding: AnalystReplayFinding, corpora: readonly CorpusSpec[]): ResolvedReplayInvocation;
|
|
755
|
+
interface ReplayFindingOptions {
|
|
756
|
+
readonly corpora: readonly CorpusSpec[];
|
|
757
|
+
readonly out: string;
|
|
758
|
+
/** Generates the arm-B corrected command; omit to run arm A only. */
|
|
759
|
+
readonly fixCaller?: ChatCompletionCaller;
|
|
760
|
+
/** Pre-supplied arm-B command; mutually exclusive with fixCaller. */
|
|
761
|
+
readonly fixCommand?: string;
|
|
762
|
+
readonly stepTimeoutMs?: number;
|
|
763
|
+
readonly prefixLimit?: number;
|
|
764
|
+
/** Builds the exec backend for the resolved trajectory image. */
|
|
765
|
+
readonly backendFactory: ReplayExecBackendFactory;
|
|
766
|
+
readonly onProgress?: (message: string) => void;
|
|
767
|
+
}
|
|
768
|
+
interface ReplayFindingResult {
|
|
769
|
+
readonly invocation: ResolvedReplayInvocation;
|
|
770
|
+
readonly fixCommand: string | null;
|
|
771
|
+
readonly verdict: ReplayVerdict;
|
|
772
|
+
}
|
|
773
|
+
/**
|
|
774
|
+
* Finding in, executed proof out. The image comes from the corpus resources;
|
|
775
|
+
* the backend factory receives it as-is, so a factory backed by infrastructure
|
|
776
|
+
* that needs a derived image must run an `ImagePreparer` first.
|
|
777
|
+
*/
|
|
778
|
+
declare function replayVerifyFinding(finding: AnalystReplayFinding, options: ReplayFindingOptions): Promise<ReplayFindingResult>;
|
|
779
|
+
//#endregion
|
|
780
|
+
export { type AnalystReplayFinding, type ArmExecutionResult, type CaseResources, type ChatCompletionCaller, type ChatOutcome, type ChatUsage, type CorpusSpec, type CwdSource, type DockerImagePreparerOptions, type EnumerationResult, type ExcludedCase, type FailedFixAttempt, type FindingReplaySource, type FindingReplayability, type FindingVerification, type FindingVerificationStatus, type FixArmExecution, type FixArmExecutor, type FixGenerationOutcome, type FixLoopAttemptRecord, type FixLoopOptions, type FixLoopResult, type FixPromptInput, type ImagePreparation, type ImagePreparer, type IncorrectStepsSubject, type IngestedTrajectory, PREFIX_DIVERGENCE_TOLERANCE_PCT, type PrefixDivergence, type PrefixDivergenceKind, type PrefixReplayResult, type RecordedTrajectoryStep, type ReplayArmVerdict, type ReplayBatchCaseRow, type ReplayBatchFixResult, type ReplayBatchOptions, type ReplayBatchReport, type ReplayExclusionReason, type ReplayExecBackend, type ReplayExecBackendFactory, type ReplayExecResult, type ReplayExecSession, type ReplayFindingOptions, type ReplayFindingResult, type ReplayVerdict, type ReplayVerifyOptions, type ReplayableCase, type ResolvedFindingReplay, type ResolvedReplayInvocation, type ResourceResolution, SUBMIT_ACTION_SIGNATURE, SandboxCounterfactualRunner, type SandboxCounterfactualRunnerOptions, type VerifiableFinding, type VerifyFindingsOptions, type VerifyFindingsRun, buildFixPrompt, buildRetryFixPrompt, classifyPrefixStep, classifyVerdict, clipText, countScriptCommands, deriveFailureSignature, derivedImageTag, dockerImagePreparer, enumerateReplayableCases, extractFixCommand, findingReplayStep, findingTrajectoryId, generateFixCommand, goldIncorrectSteps, ingestRecordedTrajectory, isSubmitAction, parseCorpusFlag, parseIncorrectStepsSubject, parseObservationOutput, parseRecordedReturncode, readFindingsFile, readLabelEntries, renderBatchReport, renderVerifiedFindingsSection, replayVerify, replayVerifyFinding, resolveCaseResources, resolveFindingInvocation, resolveFindingReplayability, runFixLoop, runReplayBatch, seededSample, summarizePrefixReplay, verifyFindings, wrapActionForExec };
|
|
781
|
+
//# sourceMappingURL=index.d.ts.map
|