@tangle-network/agent-eval 0.144.5 → 0.144.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -0
- package/dist/{benchmark-Fmo42QVE.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +471 -89
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CA_NFOmy.js → benchmark-command-BCafwNrf.js} +664 -605
- package/dist/benchmark-command-BCafwNrf.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CWbsj0t4.js → benchmarks-CDSolHq7.js} +4 -4
- package/dist/{benchmarks-CWbsj0t4.js.map → benchmarks-CDSolHq7.js.map} +1 -1
- package/dist/campaign/index.d.ts +8 -6
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-ClpnD7Ug.js → campaign-Tdy3h62h.js} +17 -301
- package/dist/campaign-Tdy3h62h.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-DAb7MWtL.d.ts → client-DjXROWpx.d.ts} +4 -4
- package/dist/{client-DAb7MWtL.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-VvpHRu78.d.ts → completion-verifier-foUCLif_.d.ts} +6 -6
- package/dist/{completion-verifier-VvpHRu78.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +11 -10
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
- package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
- package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
- package/dist/{default-registry-BwbZ9N9v.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-RLNNoeEP.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-RLNNoeEP.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-ByZiqpVk.d.ts → engine-nB64f48I.d.ts} +21 -34
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
- package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
- package/dist/errors-D-LKuDhb.js.map +1 -1
- package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-BQ7W90C4.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-BQ7W90C4.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts → feedback-trajectory-Rh280oXo.d.ts} +4 -4
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-CsuAo2-J.d.ts → index-C5HOo4ZF2.d.ts} +5 -5
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
- package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
- package/dist/{index-CNOCxBLh.d.ts → index-CvXXlyz7.d.ts} +3 -3
- package/dist/{index-CNOCxBLh.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-DGIzNtRv.d.ts → index-CwDrUMe0.d.ts} +3 -3
- package/dist/{index-DGIzNtRv.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-CpxZSlB7.d.ts → index-Sh2I0DRc.d.ts} +15 -649
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +252 -451
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +271 -751
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-C6h6F_4L.d.ts} +4 -4
- package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-DRXobPEs.d.ts → integrity-BuqEKu-x.d.ts} +3 -3
- package/dist/{integrity-DRXobPEs.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/{llm-client-D3EoChAU.js → llm-client-DzvMUsS_.js} +310 -10
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +3 -3
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-BZRWdq_t.d.ts → release-report-CI8uisI1.d.ts} +4 -4
- package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-B7S7Pdbw.d.ts → replay-DFf-teiC.d.ts} +8 -7
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +2 -2
- package/dist/{researcher-BSiCoM1s.d.ts → researcher-BoaxeCzP.d.ts} +6 -6
- package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-Cf1PtEOz.d.ts} +34 -4
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +19 -9
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
- package/dist/{run-evidence-j5Ynww6L.d.ts → run-evidence-BDFFai9R.d.ts} +3 -3
- package/dist/{run-evidence-j5Ynww6L.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
- package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DKRtp2sY.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DKRtp2sY.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-Df00sdwz.js → server-iu0ede49.js} +2 -2
- package/dist/{server-Df00sdwz.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-DUvvudWR.d.ts → skill-usage-CJlWEUFt.d.ts} +11 -11
- package/dist/{skill-usage-DUvvudWR.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-CGz9ywhM.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +11 -297
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
- package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-B4u_CiFd.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-DuUS_i7W.d.ts} +4 -115
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-BjMFz88h.d.ts → types-D216SgwM.d.ts} +228 -9
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-CZt1PBIk.d.ts → types-DF_Udrp-.d.ts} +54 -5
- package/dist/{types-CZt1PBIk.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-Dcoaqcsc.d.ts → types-DYuNHo9R.d.ts} +5 -5
- package/dist/{types-Dcoaqcsc.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/building-doctrine.md +15 -0
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +21 -4
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-Fmo42QVE.d.ts.map +0 -1
- package/dist/benchmark-command-CA_NFOmy.js.map +0 -1
- package/dist/campaign-ClpnD7Ug.js.map +0 -1
- package/dist/default-registry-BwbZ9N9v.d.ts.map +0 -1
- package/dist/index-CpxZSlB7.d.ts.map +0 -1
- package/dist/index-CsuAo2-J.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-D3EoChAU.js.map +0 -1
- package/dist/replay-B7S7Pdbw.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-DFgkEY4p.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-C4FX42dy.js.map +0 -1
- package/dist/skillopt-optimization-method-CGz9ywhM.d.ts.map +0 -1
- package/dist/summary-report-BOM6dfP7.d.ts.map +0 -1
- package/dist/tool-groups-ByZiqpVk.d.ts.map +0 -1
- package/dist/types-BjMFz88h.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
|
@@ -1,16 +1,17 @@
|
|
|
1
1
|
import { c as ValidationError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
|
|
2
2
|
import { s as resolveModelPricing } from "./metrics-C9YY1OcL.js";
|
|
3
3
|
import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-DMFxsLKr.js";
|
|
4
|
-
import { c as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-
|
|
5
|
-
import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-BHIgPmzS.js";
|
|
4
|
+
import { c as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-DzvMUsS_.js";
|
|
5
|
+
import { D as RawAnalystFindingSchema, O as evidenceRefsFromRawFinding, T as RAW_FINDING_SCHEMA_PROMPT, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-BHIgPmzS.js";
|
|
6
6
|
import { o as makeFinding, r as usageReceiptFromCostLedger } from "./usage-receipt-EVI8B8Xu.js";
|
|
7
7
|
import { i as hashCanonical, r as canonicalString } from "./canonical-D011XM8r.js";
|
|
8
|
-
import {
|
|
9
|
-
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-
|
|
8
|
+
import { D as resolveExternalOptimizerProcessLimits, f as runWithCleanup, n as createRunCostLedger, p as startExternalOptimizerModelProxy, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-BMQEv1wG.js";
|
|
9
|
+
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-BiN49gK6.js";
|
|
10
10
|
import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-DXZIqu17.js";
|
|
11
11
|
import { E as pairedBootstrap } from "./statistics-ByxzSiOM.js";
|
|
12
12
|
import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-CKtTpRhv.js";
|
|
13
13
|
import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-B181aMF9.js";
|
|
14
|
+
import { c as primeProtocolSha256, d as runPrimeExchange, f as assertEqualDeclarativeTerms, n as buildPrimePrompt, t as analystUsageReceiptFromPrimeUsage, u as projectPrimeTrajectory } from "./prime-protocol-BfSalTfR.js";
|
|
14
15
|
import { z } from "zod";
|
|
15
16
|
import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
|
|
16
17
|
import * as nodePath from "node:path";
|
|
@@ -1207,7 +1208,7 @@ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
|
|
|
1207
1208
|
"package.json",
|
|
1208
1209
|
"pnpm-lock.yaml"
|
|
1209
1210
|
]);
|
|
1210
|
-
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
1211
|
+
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "96a42e1ad7cc0c9b00b2430ae09092c14bd5de385c696e849cb50507b9d77f02";
|
|
1211
1212
|
/** The published benchmark evidence was produced at this package version, by
|
|
1212
1213
|
* the retired one-shot direct runner, before trace analysts moved to the
|
|
1213
1214
|
* recursive DSPy RLM engine. Both evidence digests below are historical facts
|
|
@@ -1252,8 +1253,10 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1252
1253
|
"src/analyst/benchmark-verification-artifacts.ts",
|
|
1253
1254
|
"src/analyst/benchmark-verification-outcome.ts",
|
|
1254
1255
|
"src/analyst/benchmark.ts",
|
|
1256
|
+
"src/analyst/definition.ts",
|
|
1255
1257
|
"src/analyst/dspy-rlm-engine.ts",
|
|
1256
1258
|
"src/analyst/engine.ts",
|
|
1259
|
+
"src/analyst/equal-terms.ts",
|
|
1257
1260
|
"src/analyst/exact-types.ts",
|
|
1258
1261
|
"src/analyst/finding-signature.ts",
|
|
1259
1262
|
"src/analyst/finding-subject.ts",
|
|
@@ -1261,6 +1264,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1261
1264
|
"src/analyst/parse-tolerant.ts",
|
|
1262
1265
|
"src/analyst/prime-bridge-transport.ts",
|
|
1263
1266
|
"src/analyst/prime-protocol.ts",
|
|
1267
|
+
"src/analyst/reply-contract.ts",
|
|
1264
1268
|
"src/analyst/tool-groups.ts",
|
|
1265
1269
|
"src/analyst/trace-tool-callback.ts",
|
|
1266
1270
|
"src/analyst/types.ts",
|
|
@@ -1279,7 +1283,9 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1279
1283
|
"src/concurrency.ts",
|
|
1280
1284
|
"src/cost-ledger.ts",
|
|
1281
1285
|
"src/errors.ts",
|
|
1286
|
+
"src/integrity/served-model.ts",
|
|
1282
1287
|
"src/judge-calibration.ts",
|
|
1288
|
+
"src/judge-families.ts",
|
|
1283
1289
|
"src/ledger-core/atomic-file-lock.ts",
|
|
1284
1290
|
"src/ledger-core/canonical.ts",
|
|
1285
1291
|
"src/ledger-core/deep-freeze.ts",
|
|
@@ -1309,7 +1315,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1309
1315
|
"src/trace/raw-provider-sink.ts",
|
|
1310
1316
|
"src/verdict-cache.ts"
|
|
1311
1317
|
]);
|
|
1312
|
-
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
1318
|
+
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "cc354effd79c8dfc8669c230706c67c63a06e3aff013f540806361263bae8860";
|
|
1313
1319
|
function analystBenchmarkImplementationDigest() {
|
|
1314
1320
|
return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256;
|
|
1315
1321
|
}
|
|
@@ -2236,19 +2242,24 @@ Cite the block's first step and its last step as trace://<URL-encoded-trace-id>/
|
|
|
2236
2242
|
Give the rationale as the concrete downstream evidence visible at the consequence step.
|
|
2237
2243
|
Submit as soon as every candidate failure block has a supported verdict.
|
|
2238
2244
|
Return no finding for a clean trajectory.`;
|
|
2239
|
-
/**
|
|
2240
|
-
|
|
2241
|
-
const fieldContract = dataset === "agentrx" ? AGENT_RX_JSON_CONTRACT : CODE_TRACE_JSON_CONTRACT;
|
|
2242
|
-
return `${publicBenchmarkTaskPrompt(dataset)}
|
|
2243
|
-
|
|
2244
|
-
${fieldContract}
|
|
2245
|
-
|
|
2246
|
-
Return exactly one JSON object with:
|
|
2245
|
+
/** Reply-envelope contract shared by both one-shot datasets. */
|
|
2246
|
+
const PUBLIC_BENCHMARK_ENVELOPE_CONTRACT = `Return exactly one JSON object with:
|
|
2247
2247
|
- "report": a concise evidence-based explanation, at most 4000 characters
|
|
2248
2248
|
- "findings": the strict finding array
|
|
2249
2249
|
Use an empty findings array when the trace does not support a finding.
|
|
2250
2250
|
Do not return a bare array, markdown, trace URIs, copied excerpts, or fields not listed above.
|
|
2251
2251
|
The runner constructs exact trace URIs and action previews from each selected step.`;
|
|
2252
|
+
/** Per-dataset field grammar for the one-shot JSON reply. */
|
|
2253
|
+
function publicBenchmarkFieldContract(dataset) {
|
|
2254
|
+
return dataset === "agentrx" ? AGENT_RX_JSON_CONTRACT : CODE_TRACE_JSON_CONTRACT;
|
|
2255
|
+
}
|
|
2256
|
+
/** One-shot JSON transport prompt for the direct runner. */
|
|
2257
|
+
function publicBenchmarkSystemPrompt(dataset) {
|
|
2258
|
+
return [
|
|
2259
|
+
publicBenchmarkTaskPrompt(dataset),
|
|
2260
|
+
publicBenchmarkFieldContract(dataset),
|
|
2261
|
+
PUBLIC_BENCHMARK_ENVELOPE_CONTRACT
|
|
2262
|
+
].join("\n\n");
|
|
2252
2263
|
}
|
|
2253
2264
|
/** Tool-loop prompt for the recursive runner. Same task, subject-encoded block. */
|
|
2254
2265
|
function publicBenchmarkRlmInstructions(dataset) {
|
|
@@ -2256,6 +2267,7 @@ function publicBenchmarkRlmInstructions(dataset) {
|
|
|
2256
2267
|
return `${publicBenchmarkTaskPrompt(dataset)}
|
|
2257
2268
|
${outputContract}`;
|
|
2258
2269
|
}
|
|
2270
|
+
/** Task text shared by every runner shape on one dataset. */
|
|
2259
2271
|
function publicBenchmarkTaskPrompt(dataset) {
|
|
2260
2272
|
return dataset === "agentrx" ? AGENT_RX_PROMPT : CODE_TRACE_BENCH_ANALYST_PROMPT;
|
|
2261
2273
|
}
|
|
@@ -3592,61 +3604,300 @@ function fileContext() {
|
|
|
3592
3604
|
};
|
|
3593
3605
|
}
|
|
3594
3606
|
//#endregion
|
|
3607
|
+
//#region src/analyst/definition.ts
|
|
3608
|
+
/**
|
|
3609
|
+
* AnalystDefinition — the declarative unit behind an analyst arm.
|
|
3610
|
+
*
|
|
3611
|
+
* An arm is one way of EXECUTING an analysis question: a one-shot JSON call, a
|
|
3612
|
+
* bridge-reached RLM, a recursive engine with trace tools. What the arm SAYS —
|
|
3613
|
+
* the question, the task text, the reply grammar, how evidence reaches the
|
|
3614
|
+
* model, the repair-turn and budget terms — is protocol, not execution, so it
|
|
3615
|
+
* lives here as one inspectable value. `bindAnalyst` (./bind) compiles a
|
|
3616
|
+
* definition plus a transport binding into a runnable arm, and the parity
|
|
3617
|
+
* suite holds the compiled arm to the byte against the arm's entry point, so a
|
|
3618
|
+
* definition cannot drift from what its arm actually sends.
|
|
3619
|
+
*
|
|
3620
|
+
* Three rules carried over from the repair-arm comparison contract
|
|
3621
|
+
* (trace-repair's `repairArmAsymmetries`), made structural here:
|
|
3622
|
+
*
|
|
3623
|
+
* one contract the reply grammar is a `ReplyContract` value on the
|
|
3624
|
+
* definition, never prose inside a runner body.
|
|
3625
|
+
* one repair turn `analystDefinitionAsymmetries` refuses a set whose
|
|
3626
|
+
* definitions declare unequal repair turns, because a second
|
|
3627
|
+
* attempt is a second sample the other arms never got.
|
|
3628
|
+
* declared difference what arms MAY differ in — the evidence projection, the
|
|
3629
|
+
* reasoning effort, the budget — is declared per definition
|
|
3630
|
+
* and rendered beside the comparison instead of being
|
|
3631
|
+
* inferred from two runners' source.
|
|
3632
|
+
*/
|
|
3633
|
+
/**
|
|
3634
|
+
* Thrown at bind time when a definition asks for something no strategy can
|
|
3635
|
+
* compile — an unknown projection × transport pair, a repair-turn count the
|
|
3636
|
+
* exchange machinery cannot grant, a reasoning effort the arm cannot map. The
|
|
3637
|
+
* message names the construct so an expressiveness gap is a loud, attributable
|
|
3638
|
+
* failure instead of a silently narrowed protocol.
|
|
3639
|
+
*/
|
|
3640
|
+
var AnalystExpressivenessError = class extends Error {};
|
|
3641
|
+
/**
|
|
3642
|
+
* Digest of everything a definition can send to its model. An inline
|
|
3643
|
+
* definition hashes under the historical prime-protocol domain, so its digest
|
|
3644
|
+
* equals the digest its bespoke arm always recorded; other projections hash
|
|
3645
|
+
* under the definition domain.
|
|
3646
|
+
*/
|
|
3647
|
+
function analystDefinitionProtocolSha256(definition) {
|
|
3648
|
+
const { projection, replyContract } = definition;
|
|
3649
|
+
if (projection.mode === "inline") return primeProtocolSha256({
|
|
3650
|
+
question: definition.question,
|
|
3651
|
+
...definition.taskDefinition === void 0 ? {} : { taskDefinition: definition.taskDefinition },
|
|
3652
|
+
contractLines: replyContract.contractLines,
|
|
3653
|
+
repairContractLines: replyContract.repairContractLines,
|
|
3654
|
+
limits: {
|
|
3655
|
+
...definition.contractLimits,
|
|
3656
|
+
maxInlineTrajectoryChars: projection.maxInlineChars,
|
|
3657
|
+
chunkedProjectionAttributeByteCap: projection.cappedAttributeBytes
|
|
3658
|
+
}
|
|
3659
|
+
});
|
|
3660
|
+
return createHash("sha256").update(JSON.stringify({
|
|
3661
|
+
kind: "analyst-definition-protocol",
|
|
3662
|
+
mode: projection.mode,
|
|
3663
|
+
question: definition.question,
|
|
3664
|
+
taskDefinition: definition.taskDefinition ?? null,
|
|
3665
|
+
contractLines: replyContract.contractLines,
|
|
3666
|
+
repairContractLines: replyContract.repairContractLines,
|
|
3667
|
+
limits: definition.contractLimits,
|
|
3668
|
+
projection: projection.mode === "chunked" ? { attributeByteCaps: projection.attributeByteCaps } : { toolGroup: projection.toolGroup }
|
|
3669
|
+
})).digest("hex");
|
|
3670
|
+
}
|
|
3671
|
+
/**
|
|
3672
|
+
* Refuse a set of definitions that cannot be compared on equal terms, and
|
|
3673
|
+
* render what still differs between the ones that can. The hard rule is the
|
|
3674
|
+
* repair turn: a malformed reply must earn the same number of retries in every
|
|
3675
|
+
* arm, because a retry is a second sample. Projection, reasoning effort, and
|
|
3676
|
+
* budget differences are declared and reported, never hidden.
|
|
3677
|
+
*/
|
|
3678
|
+
function analystDefinitionAsymmetries(definitions) {
|
|
3679
|
+
const { ids, repairTurns } = assertEqualDeclarativeTerms("analyst definition", definitions.map((definition) => ({
|
|
3680
|
+
id: definition.id,
|
|
3681
|
+
repairTurns: definition.repair.turns
|
|
3682
|
+
})));
|
|
3683
|
+
const firstMode = definitions[0].projection.mode;
|
|
3684
|
+
return {
|
|
3685
|
+
ids,
|
|
3686
|
+
repairTurns,
|
|
3687
|
+
sharedProjectionMode: definitions.every((definition) => definition.projection.mode === firstMode) ? firstMode : null,
|
|
3688
|
+
asymmetries: definitions.map((definition) => ({
|
|
3689
|
+
id: definition.id,
|
|
3690
|
+
projectionMode: definition.projection.mode,
|
|
3691
|
+
reasoningEffort: definition.profile.model?.reasoningEffort ?? null,
|
|
3692
|
+
timeoutMs: definition.budget.timeoutMs,
|
|
3693
|
+
maxCostUsd: definition.budget.maxCostUsd ?? null,
|
|
3694
|
+
maxOutputTokens: definition.budget.maxOutputTokens ?? null,
|
|
3695
|
+
protocolSha256: definition.protocolSha256,
|
|
3696
|
+
definitionSha256: analystDefinitionProtocolSha256(definition)
|
|
3697
|
+
}))
|
|
3698
|
+
};
|
|
3699
|
+
}
|
|
3700
|
+
//#endregion
|
|
3701
|
+
//#region src/analyst/reply-contract.ts
|
|
3702
|
+
/**
|
|
3703
|
+
* The reply grammar an analyst arm holds a model to, independent of transport.
|
|
3704
|
+
*
|
|
3705
|
+
* One contract serves every arm shape: the inline bridge protocol
|
|
3706
|
+
* (`runPrimeExchange`) reads the base fields, and one-shot JSON arms
|
|
3707
|
+
* additionally use the strict-envelope and all-rejected knobs. `PrimeReplyContract`
|
|
3708
|
+
* in ./prime-protocol is a type alias of this contract, so a consumer written
|
|
3709
|
+
* against the prime protocol names the same grammar object.
|
|
3710
|
+
*/
|
|
3711
|
+
/**
|
|
3712
|
+
* Decode a parsed reply value under a contract: strict envelope when declared,
|
|
3713
|
+
* then per-row decoding, then the all-rejected policy. Shape before count: a
|
|
3714
|
+
* malformed row never consumes an accepted slot.
|
|
3715
|
+
*/
|
|
3716
|
+
function decodeReplyRows(contract, value) {
|
|
3717
|
+
let rawRows;
|
|
3718
|
+
let extras = {};
|
|
3719
|
+
if (contract.parseEnvelope) {
|
|
3720
|
+
const envelope = contract.parseEnvelope(value);
|
|
3721
|
+
rawRows = envelope.rows;
|
|
3722
|
+
extras = envelope.extras;
|
|
3723
|
+
} else {
|
|
3724
|
+
const field = (typeof value === "object" && value !== null && !Array.isArray(value) ? value : void 0)?.[contract.rowsField];
|
|
3725
|
+
if (!Array.isArray(field)) throw new ValidationError(`reply has no "${contract.rowsField}" array`);
|
|
3726
|
+
rawRows = field;
|
|
3727
|
+
}
|
|
3728
|
+
const rows = [];
|
|
3729
|
+
const rejected = [];
|
|
3730
|
+
let overflow = 0;
|
|
3731
|
+
rawRows.forEach((row, index) => {
|
|
3732
|
+
const decoded = contract.decodeRow(row, index);
|
|
3733
|
+
if (!decoded.ok) {
|
|
3734
|
+
rejected.push({
|
|
3735
|
+
index,
|
|
3736
|
+
reason: decoded.reason
|
|
3737
|
+
});
|
|
3738
|
+
return;
|
|
3739
|
+
}
|
|
3740
|
+
if (contract.maxRows !== void 0 && rows.length >= contract.maxRows) {
|
|
3741
|
+
overflow += 1;
|
|
3742
|
+
return;
|
|
3743
|
+
}
|
|
3744
|
+
rows.push(decoded.row);
|
|
3745
|
+
});
|
|
3746
|
+
if (rows.length === 0 && rawRows.length > 0 && contract.whenAllRowsRejected === "fail") throw new ValidationError(`${contract.allRejectedMessage ?? "every reported row was malformed"}: ${rejected.map((entry) => entry.reason).join(" | ")}`);
|
|
3747
|
+
return {
|
|
3748
|
+
rows,
|
|
3749
|
+
extras,
|
|
3750
|
+
rejected,
|
|
3751
|
+
reportedRows: rawRows.length,
|
|
3752
|
+
overflow
|
|
3753
|
+
};
|
|
3754
|
+
}
|
|
3755
|
+
//#endregion
|
|
3595
3756
|
//#region src/analyst/benchmark-public-model.ts
|
|
3596
|
-
/**
|
|
3757
|
+
/** The direct arm as a declarative unit for one public dataset. */
|
|
3758
|
+
function publicDirectAnalystDefinition(dataset, args) {
|
|
3759
|
+
const actor = dataset === "agentrx" ? "agentrx-root-cause-localizer" : "codetracebench-step-localizer";
|
|
3760
|
+
const outputAdapter = dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-block";
|
|
3761
|
+
return {
|
|
3762
|
+
id: "direct",
|
|
3763
|
+
description: "One-shot JSON baseline over the caller-owned model path.",
|
|
3764
|
+
version: "1.0.0",
|
|
3765
|
+
area: dataset === "agentrx" ? "root-cause" : "incorrect",
|
|
3766
|
+
profile: { model: { reasoningEffort: "none" } },
|
|
3767
|
+
question: "",
|
|
3768
|
+
taskDefinition: publicBenchmarkTaskPrompt(dataset),
|
|
3769
|
+
projection: {
|
|
3770
|
+
mode: "chunked",
|
|
3771
|
+
attributeByteCaps: TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS
|
|
3772
|
+
},
|
|
3773
|
+
replyContract: directReplyContract(dataset),
|
|
3774
|
+
contractLimits: dataset === "agentrx" ? { maxFindings: 1 } : {
|
|
3775
|
+
maxBlocks: 16,
|
|
3776
|
+
maxBlockSteps: 12
|
|
3777
|
+
},
|
|
3778
|
+
budget: {
|
|
3779
|
+
timeoutMs: args.timeoutMs,
|
|
3780
|
+
maxCostUsd: args.maxCostUsd,
|
|
3781
|
+
maxOutputTokens: args.maxOutputTokens
|
|
3782
|
+
},
|
|
3783
|
+
repair: { turns: 0 },
|
|
3784
|
+
protocolSha256: publicBenchmarkProtocolSha256(dataset),
|
|
3785
|
+
binding: {
|
|
3786
|
+
kind: "chunked",
|
|
3787
|
+
subjectFromCaseId: (caseId) => trajectoryIdFromCaseId$2(dataset, caseId),
|
|
3788
|
+
baseMetadata: {
|
|
3789
|
+
analysisMode: "direct-baseline",
|
|
3790
|
+
outputAdapter
|
|
3791
|
+
},
|
|
3792
|
+
costActor: actor,
|
|
3793
|
+
costPhase: "analyst.public-benchmark",
|
|
3794
|
+
userMessage: (rendered) => `TRACE DATA:\n${rendered}\n\nReturn the analysis JSON object.`,
|
|
3795
|
+
async expandRows({ subject, rows, store, analystId, producedAt, providerModel, signal }) {
|
|
3796
|
+
const converted = await publicBenchmarkPredictionsToFindings({
|
|
3797
|
+
dataset,
|
|
3798
|
+
trajectoryId: subject,
|
|
3799
|
+
predictions: rows,
|
|
3800
|
+
store,
|
|
3801
|
+
analystId,
|
|
3802
|
+
providerModel: requiredString(providerModel ?? "", "finding providerModel"),
|
|
3803
|
+
producedAt: requiredString(producedAt ?? "", "finding producedAt"),
|
|
3804
|
+
...signal ? { signal } : {}
|
|
3805
|
+
});
|
|
3806
|
+
return {
|
|
3807
|
+
findings: converted.findings,
|
|
3808
|
+
diagnostics: converted.diagnostics
|
|
3809
|
+
};
|
|
3810
|
+
},
|
|
3811
|
+
...dataset === "codetracebench" ? { verifyFindings: async (args) => {
|
|
3812
|
+
await validateCodeTraceFindingEvidence({
|
|
3813
|
+
trajectoryId: args.subject,
|
|
3814
|
+
findings: [...args.findings],
|
|
3815
|
+
store: args.store,
|
|
3816
|
+
...args.signal ? { signal: args.signal } : {}
|
|
3817
|
+
});
|
|
3818
|
+
} } : {}
|
|
3819
|
+
}
|
|
3820
|
+
};
|
|
3821
|
+
}
|
|
3822
|
+
/** Thin shell: validate config, declare the definition, run the chunked strategy. */
|
|
3597
3823
|
function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
3598
3824
|
if (config.instructionsOverride) throw new Error("the direct runner executes only the stock protocol; an instructions override requires the dspy-rlm runner");
|
|
3825
|
+
const maxOutputTokens = positiveSafeInteger(config.maxOutputTokens, "maxOutputTokens");
|
|
3826
|
+
return runChunkedAnalystDefinition(publicDirectAnalystDefinition(dataset, {
|
|
3827
|
+
timeoutMs: positiveSafeInteger(config.timeoutMs, "timeoutMs"),
|
|
3828
|
+
maxOutputTokens,
|
|
3829
|
+
maxCostUsd: config.maxCostUsdPerAnalysis ?? 1
|
|
3830
|
+
}), config);
|
|
3831
|
+
}
|
|
3832
|
+
/**
|
|
3833
|
+
* Compile a chunked-projection definition into a runnable one-shot JSON arm
|
|
3834
|
+
* over the caller-owned model path. Prompt content, the projection ladder,
|
|
3835
|
+
* the reply grammar, and the budget declaration come from the definition;
|
|
3836
|
+
* caching, cost settlement, and the model proxy are transport machinery.
|
|
3837
|
+
*/
|
|
3838
|
+
function runChunkedAnalystDefinition(definition, config) {
|
|
3839
|
+
const { projection, binding, replyContract } = definition;
|
|
3840
|
+
if (projection.mode !== "chunked" || binding.kind !== "chunked") throw new AnalystExpressivenessError(`the chunked one-shot strategy compiles only chunked projections; definition '${definition.id}' declares projection '${projection.mode}' with a '${binding.kind}' binding`);
|
|
3841
|
+
if (config.instructionsOverride) throw new Error("the direct runner executes only the stock protocol; an instructions override requires the dspy-rlm runner");
|
|
3842
|
+
if (definition.repair.turns !== 0) throw new AnalystExpressivenessError(`the one-shot JSON exchange grants no repair turn; definition '${definition.id}' declares ${definition.repair.turns}`);
|
|
3843
|
+
const reasoningEffort = definition.profile.model?.reasoningEffort;
|
|
3844
|
+
if (reasoningEffort !== "none") throw new AnalystExpressivenessError(`the one-shot JSON strategy runs with thinking disabled and can express only reasoning effort 'none'; definition '${definition.id}' declares '${reasoningEffort}'`);
|
|
3845
|
+
if (!replyContract.parseEnvelope) throw new AnalystExpressivenessError(`the one-shot JSON strategy needs a strict reply envelope; definition '${definition.id}' declares no parseEnvelope`);
|
|
3846
|
+
if (definition.taskDefinition === void 0) throw new AnalystExpressivenessError(`the one-shot JSON strategy composes its system prompt from the task definition; definition '${definition.id}' declares none`);
|
|
3599
3847
|
const model = requiredString(config.model, "model");
|
|
3600
3848
|
const callRef = requiredString(config.callRef, "callRef");
|
|
3601
3849
|
if (typeof config.call !== "function") throw new TypeError("call must be a function");
|
|
3602
3850
|
if (typeof config.recordExecution !== "function") throw new TypeError("recordExecution must be a function");
|
|
3603
3851
|
const maxOutputTokens = positiveSafeInteger(config.maxOutputTokens, "maxOutputTokens");
|
|
3604
3852
|
const timeoutMs = positiveSafeInteger(config.timeoutMs, "timeoutMs");
|
|
3853
|
+
const maxCostUsd = config.maxCostUsdPerAnalysis ?? 1;
|
|
3854
|
+
assertDeclaredBudget(definition, {
|
|
3855
|
+
timeoutMs,
|
|
3856
|
+
maxOutputTokens,
|
|
3857
|
+
maxCostUsd
|
|
3858
|
+
});
|
|
3605
3859
|
const maxReasoningTokens = config.maxReasoningTokens ?? maxOutputTokens * 4;
|
|
3606
3860
|
const maxModelRequestBytes = config.maxModelRequestBytes ?? 16 * 1024 * 1024;
|
|
3607
3861
|
const maxModelResponseBytes = config.maxModelResponseBytes ?? 4 * 1024 * 1024;
|
|
3608
3862
|
const modelRequestTimeoutMs = config.modelRequestTimeoutMs ?? timeoutMs;
|
|
3609
3863
|
const pricing = config.pricing ?? pricingForModel$2(model);
|
|
3610
|
-
const maxCostUsd = config.maxCostUsdPerAnalysis ?? 1;
|
|
3611
3864
|
const costLedger = config.costLedger ?? new CostLedger();
|
|
3612
3865
|
const durability = config.durability ? {
|
|
3613
3866
|
runIdentitySha256: requiredString(config.durability.runIdentitySha256, "durability.runIdentitySha256"),
|
|
3614
3867
|
responseCacheDir: requiredString(config.durability.responseCacheDir, "durability.responseCacheDir")
|
|
3615
3868
|
} : void 0;
|
|
3616
|
-
const
|
|
3617
|
-
const outputAdapter = dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-block";
|
|
3869
|
+
const systemPrompt = [definition.taskDefinition, ...replyContract.contractLines].join("\n\n");
|
|
3618
3870
|
return {
|
|
3619
|
-
id:
|
|
3871
|
+
id: definition.id,
|
|
3620
3872
|
async analyze(input, context) {
|
|
3621
|
-
const trajectoryId =
|
|
3873
|
+
const trajectoryId = binding.subjectFromCaseId(context.caseId);
|
|
3622
3874
|
const costTags = {
|
|
3623
|
-
analystId:
|
|
3875
|
+
analystId: binding.costActor,
|
|
3624
3876
|
benchmarkCaseId: context.caseId,
|
|
3625
3877
|
benchmarkRepetition: String(context.repetition)
|
|
3626
3878
|
};
|
|
3627
3879
|
let rawPredictions = [];
|
|
3628
|
-
let
|
|
3880
|
+
let rejectedRows = [];
|
|
3629
3881
|
let modelFindings = [];
|
|
3630
3882
|
let providerModel = model;
|
|
3631
3883
|
let producedAt;
|
|
3632
3884
|
let modelMetadata = {
|
|
3633
|
-
|
|
3634
|
-
|
|
3635
|
-
protocolSha256: publicBenchmarkProtocolSha256(dataset),
|
|
3885
|
+
...binding.baseMetadata,
|
|
3886
|
+
protocolSha256: definition.protocolSha256,
|
|
3636
3887
|
callRef
|
|
3637
3888
|
};
|
|
3638
3889
|
try {
|
|
3639
|
-
if (!input.traceStore) throw new Error(
|
|
3640
|
-
const preparedContext = await prepareSingleTraceContext(input.traceStore, context);
|
|
3641
|
-
if (preparedContext === void 0) throw new Error(
|
|
3890
|
+
if (!input.traceStore) throw new Error(`chunked analyst '${definition.id}' requires a trace store`);
|
|
3891
|
+
const preparedContext = await prepareSingleTraceContext(input.traceStore, context, projection.attributeByteCaps);
|
|
3892
|
+
if (preparedContext === void 0) throw new Error(`trace '${trajectoryId}' has no readable spans`);
|
|
3642
3893
|
const request = {
|
|
3643
3894
|
model,
|
|
3644
3895
|
messages: [{
|
|
3645
3896
|
role: "system",
|
|
3646
|
-
content:
|
|
3897
|
+
content: systemPrompt
|
|
3647
3898
|
}, {
|
|
3648
3899
|
role: "user",
|
|
3649
|
-
content:
|
|
3900
|
+
content: binding.userMessage(preparedContext)
|
|
3650
3901
|
}],
|
|
3651
3902
|
jsonMode: true,
|
|
3652
3903
|
thinking: "disabled",
|
|
@@ -3676,14 +3927,14 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
3676
3927
|
error: cached.error,
|
|
3677
3928
|
metadata: modelMetadata
|
|
3678
3929
|
};
|
|
3679
|
-
const response =
|
|
3680
|
-
rawPredictions = response.
|
|
3681
|
-
|
|
3930
|
+
const response = decodeReplyRows(replyContract, cached.response);
|
|
3931
|
+
rawPredictions = response.rows;
|
|
3932
|
+
rejectedRows = response.rejected.map((entry) => entry.reason);
|
|
3682
3933
|
providerModel = cached.metadata.providerModel;
|
|
3683
3934
|
producedAt = cached.metadata.producedAt;
|
|
3684
3935
|
modelMetadata = {
|
|
3685
3936
|
...modelMetadata,
|
|
3686
|
-
|
|
3937
|
+
...response.extras,
|
|
3687
3938
|
providerModel: cached.metadata.providerModel,
|
|
3688
3939
|
providerDurationMs: cached.metadata.providerDurationMs,
|
|
3689
3940
|
finishReason: cached.metadata.finishReason
|
|
@@ -3712,8 +3963,8 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
3712
3963
|
},
|
|
3713
3964
|
costLedger,
|
|
3714
3965
|
channel: "analyst",
|
|
3715
|
-
phase:
|
|
3716
|
-
actor,
|
|
3966
|
+
phase: binding.costPhase,
|
|
3967
|
+
actor: binding.costActor,
|
|
3717
3968
|
tags: costTags,
|
|
3718
3969
|
callId: providerCallId,
|
|
3719
3970
|
...context.signal ? { signal: context.signal } : {}
|
|
@@ -3732,7 +3983,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
3732
3983
|
...context.signal ? { signal: context.signal } : {},
|
|
3733
3984
|
idempotencyKey: providerCallId
|
|
3734
3985
|
});
|
|
3735
|
-
const response =
|
|
3986
|
+
const response = decodeReplyRows(replyContract, completed.value);
|
|
3736
3987
|
const responseProducedAt = (/* @__PURE__ */ new Date()).toISOString();
|
|
3737
3988
|
const receipt = requiredSettledReceipt(costLedger, providerCallId);
|
|
3738
3989
|
if (cacheIdentity) writePublicBenchmarkResponseCache(durability.responseCacheDir, {
|
|
@@ -3778,26 +4029,25 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
3778
4029
|
}
|
|
3779
4030
|
});
|
|
3780
4031
|
const response = completed.response;
|
|
3781
|
-
rawPredictions = response.
|
|
3782
|
-
|
|
4032
|
+
rawPredictions = response.rows;
|
|
4033
|
+
rejectedRows = response.rejected.map((entry) => entry.reason);
|
|
3783
4034
|
providerModel = completed.result.model;
|
|
3784
4035
|
producedAt = completed.producedAt;
|
|
3785
4036
|
modelMetadata = {
|
|
3786
4037
|
...modelMetadata,
|
|
3787
4038
|
responseSource: "provider",
|
|
3788
|
-
|
|
4039
|
+
...response.extras,
|
|
3789
4040
|
providerModel: completed.result.model,
|
|
3790
4041
|
providerDurationMs: completed.result.durationMs,
|
|
3791
4042
|
finishReason: completed.result.finishReason ?? null,
|
|
3792
4043
|
cost: costReceiptMetadata(completed.receipt)
|
|
3793
4044
|
};
|
|
3794
4045
|
}
|
|
3795
|
-
const converted = await
|
|
3796
|
-
|
|
3797
|
-
|
|
3798
|
-
predictions: rawPredictions,
|
|
4046
|
+
const converted = await binding.expandRows({
|
|
4047
|
+
subject: trajectoryId,
|
|
4048
|
+
rows: rawPredictions,
|
|
3799
4049
|
store: input.traceStore,
|
|
3800
|
-
analystId:
|
|
4050
|
+
analystId: definition.id,
|
|
3801
4051
|
providerModel,
|
|
3802
4052
|
producedAt: requiredString(producedAt ?? "", "finding producedAt"),
|
|
3803
4053
|
...context.signal ? { signal: context.signal } : {}
|
|
@@ -3807,11 +4057,11 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
3807
4057
|
...modelMetadata,
|
|
3808
4058
|
blockDiagnostics: {
|
|
3809
4059
|
...converted.diagnostics,
|
|
3810
|
-
rejectedBlocks
|
|
4060
|
+
rejectedBlocks: rejectedRows
|
|
3811
4061
|
}
|
|
3812
4062
|
};
|
|
3813
|
-
if (
|
|
3814
|
-
trajectoryId,
|
|
4063
|
+
if (binding.verifyFindings) await binding.verifyFindings({
|
|
4064
|
+
subject: trajectoryId,
|
|
3815
4065
|
findings: modelFindings,
|
|
3816
4066
|
store: input.traceStore,
|
|
3817
4067
|
...context.signal ? { signal: context.signal } : {}
|
|
@@ -3844,6 +4094,11 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
3844
4094
|
}
|
|
3845
4095
|
};
|
|
3846
4096
|
}
|
|
4097
|
+
/** A definition that declares one budget while the transport runs another is refused. */
|
|
4098
|
+
function assertDeclaredBudget(definition, effective) {
|
|
4099
|
+
const declared = definition.budget;
|
|
4100
|
+
if (declared.timeoutMs !== effective.timeoutMs || declared.maxOutputTokens !== effective.maxOutputTokens || declared.maxCostUsd !== effective.maxCostUsd) throw new AnalystExpressivenessError(`definition '${definition.id}' declares budget ${JSON.stringify(declared)} but the bound transport runs ${JSON.stringify(effective)}; the declaration must state what executes`);
|
|
4101
|
+
}
|
|
3847
4102
|
function settleCachedResponse(costLedger, cached) {
|
|
3848
4103
|
const settled = costLedger.list().find((receipt) => receipt.callId === cached.callId);
|
|
3849
4104
|
const pending = costLedger.listPending?.().find((record) => record.callId === cached.callId);
|
|
@@ -4001,27 +4256,56 @@ const AgentRxModelResponseSchema = z.object({
|
|
|
4001
4256
|
report: z.string().min(1).max(4e3),
|
|
4002
4257
|
findings: z.array(AgentRxPredictionSchema.extend({ category: AgentRxCategorySchema }).strict()).max(1)
|
|
4003
4258
|
}).strict();
|
|
4004
|
-
|
|
4259
|
+
/**
|
|
4260
|
+
* The one-shot reply grammar per dataset. The envelope is the contract and
|
|
4261
|
+
* stays strict. Individual CodeTraceBench blocks are model output: one
|
|
4262
|
+
* malformed block must not void a case whose remaining blocks are usable and
|
|
4263
|
+
* whose provider call is already paid for, so rows decode individually and
|
|
4264
|
+
* every rejection is reported.
|
|
4265
|
+
*/
|
|
4266
|
+
function directReplyContract(dataset) {
|
|
4005
4267
|
if (dataset === "agentrx") return {
|
|
4006
|
-
|
|
4007
|
-
|
|
4008
|
-
|
|
4009
|
-
|
|
4010
|
-
|
|
4011
|
-
|
|
4012
|
-
|
|
4013
|
-
|
|
4014
|
-
|
|
4015
|
-
|
|
4016
|
-
|
|
4268
|
+
rowsField: "findings",
|
|
4269
|
+
contractLines: [publicBenchmarkFieldContract("agentrx"), PUBLIC_BENCHMARK_ENVELOPE_CONTRACT],
|
|
4270
|
+
repairContractLines: [],
|
|
4271
|
+
parseEnvelope(value) {
|
|
4272
|
+
const parsed = AgentRxModelResponseSchema.parse(value);
|
|
4273
|
+
return {
|
|
4274
|
+
rows: parsed.findings,
|
|
4275
|
+
extras: { report: parsed.report }
|
|
4276
|
+
};
|
|
4277
|
+
},
|
|
4278
|
+
decodeRow(row) {
|
|
4279
|
+
return {
|
|
4280
|
+
ok: true,
|
|
4281
|
+
row
|
|
4282
|
+
};
|
|
4017
4283
|
}
|
|
4018
|
-
|
|
4019
|
-
}
|
|
4020
|
-
if (findings.length === 0 && envelope.findings.length > 0) throw new ValidationError(`every reported failure block was malformed: ${rejectedBlocks.join(" | ")}`);
|
|
4284
|
+
};
|
|
4021
4285
|
return {
|
|
4022
|
-
|
|
4023
|
-
|
|
4024
|
-
|
|
4286
|
+
rowsField: "findings",
|
|
4287
|
+
contractLines: [publicBenchmarkFieldContract("codetracebench"), PUBLIC_BENCHMARK_ENVELOPE_CONTRACT],
|
|
4288
|
+
repairContractLines: [],
|
|
4289
|
+
parseEnvelope(value) {
|
|
4290
|
+
const envelope = CodeTraceModelResponseEnvelopeSchema.parse(value);
|
|
4291
|
+
return {
|
|
4292
|
+
rows: envelope.findings,
|
|
4293
|
+
extras: { report: envelope.report }
|
|
4294
|
+
};
|
|
4295
|
+
},
|
|
4296
|
+
decodeRow(row, index) {
|
|
4297
|
+
const parsed = CodeTraceBlockPredictionSchema.safeParse(row);
|
|
4298
|
+
if (parsed.success) return {
|
|
4299
|
+
ok: true,
|
|
4300
|
+
row: parsed.data
|
|
4301
|
+
};
|
|
4302
|
+
return {
|
|
4303
|
+
ok: false,
|
|
4304
|
+
reason: `block ${index}: ${parsed.error.issues.map((issue) => `${issue.path.join(".") || "<root>"} ${issue.message}`).join("; ")}`
|
|
4305
|
+
};
|
|
4306
|
+
},
|
|
4307
|
+
whenAllRowsRejected: "fail",
|
|
4308
|
+
allRejectedMessage: "every reported failure block was malformed"
|
|
4025
4309
|
};
|
|
4026
4310
|
}
|
|
4027
4311
|
async function publicBenchmarkPredictionsToFindings(options) {
|
|
@@ -4088,12 +4372,12 @@ async function publicBenchmarkPredictionsToFindings(options) {
|
|
|
4088
4372
|
...options.signal ? { signal: options.signal } : {}
|
|
4089
4373
|
});
|
|
4090
4374
|
}
|
|
4091
|
-
async function prepareSingleTraceContext(store, context) {
|
|
4375
|
+
async function prepareSingleTraceContext(store, context, attributeByteCaps) {
|
|
4092
4376
|
const storeContext = context.signal ? { signal: context.signal } : void 0;
|
|
4093
4377
|
const overview = await store.getOverview(void 0, storeContext);
|
|
4094
4378
|
if (overview.total_traces !== 1 || overview.sample_trace_ids.length !== 1) throw new Error(`public model benchmark requires exactly one trace, received ${overview.total_traces}`);
|
|
4095
4379
|
const traceId = overview.sample_trace_ids[0];
|
|
4096
|
-
for (const perAttributeByteCap of
|
|
4380
|
+
for (const perAttributeByteCap of attributeByteCaps) {
|
|
4097
4381
|
const viewed = await store.viewTrace({
|
|
4098
4382
|
trace_id: traceId,
|
|
4099
4383
|
per_attribute_byte_cap: perAttributeByteCap
|
|
@@ -4243,18 +4527,156 @@ function contiguousSegments(sortedSteps) {
|
|
|
4243
4527
|
}
|
|
4244
4528
|
//#endregion
|
|
4245
4529
|
//#region src/analyst/benchmark-public-rlm.ts
|
|
4246
|
-
/**
|
|
4530
|
+
/** The dspy-rlm arm as a declarative unit for one public dataset. */
|
|
4531
|
+
function publicRlmAnalystDefinition(dataset, args) {
|
|
4532
|
+
return {
|
|
4533
|
+
id: "dspy-rlm",
|
|
4534
|
+
description: dataset === "agentrx" ? "Localizes the first unrecoverable root-cause step." : "Localizes every incorrect state-changing assistant step.",
|
|
4535
|
+
version: "1.0.0",
|
|
4536
|
+
area: dataset === "agentrx" ? "root-cause" : "incorrect",
|
|
4537
|
+
profile: {},
|
|
4538
|
+
question: dataset === "agentrx" ? "What is the first unrecoverable root cause in this failed trajectory?" : "Which assistant steps are incorrect under the CodeTraceBench definition?",
|
|
4539
|
+
taskDefinition: args.instructions,
|
|
4540
|
+
projection: {
|
|
4541
|
+
mode: "repl-variable",
|
|
4542
|
+
toolGroup: "singleTrace"
|
|
4543
|
+
},
|
|
4544
|
+
replyContract: {
|
|
4545
|
+
rowsField: "findings",
|
|
4546
|
+
contractLines: [RAW_FINDING_SCHEMA_PROMPT],
|
|
4547
|
+
repairContractLines: [],
|
|
4548
|
+
decodeRow(row) {
|
|
4549
|
+
const parsed = RawAnalystFindingSchema.safeParse(row);
|
|
4550
|
+
if (parsed.success) return {
|
|
4551
|
+
ok: true,
|
|
4552
|
+
row: parsed.data
|
|
4553
|
+
};
|
|
4554
|
+
return {
|
|
4555
|
+
ok: false,
|
|
4556
|
+
reason: parsed.error.issues.map((issue) => `${issue.path.join(".")}: ${issue.message}`).join("; ")
|
|
4557
|
+
};
|
|
4558
|
+
}
|
|
4559
|
+
},
|
|
4560
|
+
contractLimits: {
|
|
4561
|
+
maxIterations: args.engineLimits.maxIterations,
|
|
4562
|
+
maxLlmCalls: args.engineLimits.maxLlmCalls,
|
|
4563
|
+
maxToolCalls: args.engineLimits.maxToolCalls,
|
|
4564
|
+
maxOutputChars: args.engineLimits.maxOutputChars
|
|
4565
|
+
},
|
|
4566
|
+
budget: {
|
|
4567
|
+
timeoutMs: args.timeoutMs,
|
|
4568
|
+
maxCostUsd: args.maxCostUsd,
|
|
4569
|
+
maxOutputTokens: args.maxOutputTokens,
|
|
4570
|
+
engineLimits: args.engineLimits
|
|
4571
|
+
},
|
|
4572
|
+
repair: { turns: 1 },
|
|
4573
|
+
protocolSha256: args.protocolSha256,
|
|
4574
|
+
binding: {
|
|
4575
|
+
kind: "repl-variable",
|
|
4576
|
+
traceAnalystId: dataset === "agentrx" ? "agentrx-dspy-rlm" : "codetracebench-dspy-rlm",
|
|
4577
|
+
subjectFromCaseId: (caseId) => trajectoryIdFromCaseId$1(dataset, caseId),
|
|
4578
|
+
baseMetadata: {
|
|
4579
|
+
analysisMode: "recursive",
|
|
4580
|
+
engine: "dspy-rlm"
|
|
4581
|
+
},
|
|
4582
|
+
findingBaseMetadata: {
|
|
4583
|
+
analysis_mode: "recursive",
|
|
4584
|
+
engine: "dspy-rlm"
|
|
4585
|
+
},
|
|
4586
|
+
costPhase: "analyst.public-benchmark.dspy-rlm",
|
|
4587
|
+
...dataset === "codetracebench" ? { metadataFromSubject: codeTraceBlockMetadataFromSubject } : {},
|
|
4588
|
+
async adapt({ subject, findings, analystId, store, signal }) {
|
|
4589
|
+
return adaptPublicBenchmarkFindings({
|
|
4590
|
+
dataset,
|
|
4591
|
+
trajectoryId: subject,
|
|
4592
|
+
findings: [...findings],
|
|
4593
|
+
analystId,
|
|
4594
|
+
store,
|
|
4595
|
+
...signal ? { signal } : {}
|
|
4596
|
+
});
|
|
4597
|
+
},
|
|
4598
|
+
...dataset === "codetracebench" ? { consensus: codeTraceConsensusPort() } : {},
|
|
4599
|
+
abstentionFallback: (fallbackConfig) => createPublicBenchmarkDirectRunner(dataset, fallbackConfig)
|
|
4600
|
+
}
|
|
4601
|
+
};
|
|
4602
|
+
}
|
|
4603
|
+
/** Step-level majority consensus on the CodeTraceBench block grammar. */
|
|
4604
|
+
function codeTraceConsensusPort() {
|
|
4605
|
+
return {
|
|
4606
|
+
vote(samples) {
|
|
4607
|
+
const consensus = consensusCodeTraceBlocks(samples.map((sample) => [...sample]));
|
|
4608
|
+
return {
|
|
4609
|
+
blocks: consensus.blocks,
|
|
4610
|
+
decision: consensus.decision
|
|
4611
|
+
};
|
|
4612
|
+
},
|
|
4613
|
+
async expand({ subject, blocks, store, analystId, producedAt, signal }) {
|
|
4614
|
+
const expanded = await expandCodeTraceFailureBlocks({
|
|
4615
|
+
trajectoryId: subject,
|
|
4616
|
+
blocks,
|
|
4617
|
+
store,
|
|
4618
|
+
analystId,
|
|
4619
|
+
producedAt,
|
|
4620
|
+
...signal ? { signal } : {}
|
|
4621
|
+
});
|
|
4622
|
+
return {
|
|
4623
|
+
findings: expanded.findings,
|
|
4624
|
+
diagnostics: expanded.diagnostics
|
|
4625
|
+
};
|
|
4626
|
+
},
|
|
4627
|
+
sampleRecord(assignments) {
|
|
4628
|
+
return {
|
|
4629
|
+
blocks: sampleBlockRecords(assignments),
|
|
4630
|
+
steps: assignments.map((assignment) => assignment.step)
|
|
4631
|
+
};
|
|
4632
|
+
}
|
|
4633
|
+
};
|
|
4634
|
+
}
|
|
4635
|
+
/** Thin shell: validate config, declare the definition, run the repl-variable strategy. */
|
|
4247
4636
|
function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
4248
|
-
const costLedger = config.costLedger ?? new CostLedger();
|
|
4249
4637
|
const samples = config.dspyRlm?.samples ?? 1;
|
|
4250
4638
|
if (!Number.isSafeInteger(samples) || samples < 1) throw new RangeError("dspyRlm.samples must be a positive safe integer");
|
|
4251
4639
|
if (samples > 1 && dataset !== "codetracebench") throw new Error("dspyRlm.samples > 1 requires the codetracebench dataset; step-level consensus is defined on its block grammar");
|
|
4252
|
-
|
|
4640
|
+
return runReplVariableAnalystDefinition(publicRlmAnalystDefinition(dataset, {
|
|
4641
|
+
instructions: config.instructionsOverride?.text ?? publicBenchmarkRlmInstructions(dataset),
|
|
4642
|
+
protocolSha256: effectiveAnalystProtocolSha256(dataset, config.instructionsOverride),
|
|
4643
|
+
timeoutMs: config.timeoutMs,
|
|
4644
|
+
maxOutputTokens: config.maxOutputTokens,
|
|
4645
|
+
maxCostUsd: config.maxCostUsdPerAnalysis ?? 1,
|
|
4646
|
+
engineLimits: rlmEngineLimits(config)
|
|
4647
|
+
}), config);
|
|
4648
|
+
}
|
|
4649
|
+
/** Engine iteration limits with this arm's defaults applied. */
|
|
4650
|
+
function rlmEngineLimits(config) {
|
|
4651
|
+
return {
|
|
4253
4652
|
maxIterations: config.dspyRlm?.maxIterations ?? 14,
|
|
4254
4653
|
maxLlmCalls: config.dspyRlm?.maxLlmCalls ?? 8,
|
|
4255
4654
|
maxToolCalls: config.dspyRlm?.maxToolCalls ?? 80,
|
|
4256
4655
|
maxOutputChars: config.dspyRlm?.maxOutputChars ?? 8e3
|
|
4257
4656
|
};
|
|
4657
|
+
}
|
|
4658
|
+
/**
|
|
4659
|
+
* Compile a repl-variable definition into a runnable recursive-engine arm over
|
|
4660
|
+
* the caller-owned model path. The question, instructions, tool group, and
|
|
4661
|
+
* iteration limits come from the definition; the engine, model proxy, sampling
|
|
4662
|
+
* loop, and abstention floor are transport machinery.
|
|
4663
|
+
*/
|
|
4664
|
+
function runReplVariableAnalystDefinition(definition, config) {
|
|
4665
|
+
const { projection, binding } = definition;
|
|
4666
|
+
if (projection.mode !== "repl-variable" || binding.kind !== "repl-variable") throw new AnalystExpressivenessError(`the repl-variable strategy compiles only repl-variable projections; definition '${definition.id}' declares projection '${projection.mode}' with a '${binding.kind}' binding`);
|
|
4667
|
+
const instructions = definition.taskDefinition;
|
|
4668
|
+
if (instructions === void 0) throw new AnalystExpressivenessError(`the repl-variable strategy runs the definition's task text as engine instructions; definition '${definition.id}' declares none`);
|
|
4669
|
+
if (config.instructionsOverride !== void 0 && config.instructionsOverride.text !== instructions) throw new AnalystExpressivenessError(`definition '${definition.id}' declares instructions that differ from the transport's instructionsOverride; one text must execute`);
|
|
4670
|
+
const area = definition.area;
|
|
4671
|
+
if (area === void 0) throw new AnalystExpressivenessError(`the repl-variable strategy stamps the definition's area on every finding; definition '${definition.id}' declares none`);
|
|
4672
|
+
const limits = definition.budget.engineLimits;
|
|
4673
|
+
if (limits === void 0) throw new AnalystExpressivenessError(`the repl-variable strategy needs declared engine limits; definition '${definition.id}' declares none`);
|
|
4674
|
+
const effectiveLimits = rlmEngineLimits(config);
|
|
4675
|
+
if (limits.maxIterations !== effectiveLimits.maxIterations || limits.maxLlmCalls !== effectiveLimits.maxLlmCalls || limits.maxToolCalls !== effectiveLimits.maxToolCalls || limits.maxOutputChars !== effectiveLimits.maxOutputChars) throw new AnalystExpressivenessError(`definition '${definition.id}' declares engine limits ${JSON.stringify(limits)} but the bound transport runs ${JSON.stringify(effectiveLimits)}; the declaration must state what executes`);
|
|
4676
|
+
const costLedger = config.costLedger ?? new CostLedger();
|
|
4677
|
+
const samples = config.dspyRlm?.samples ?? 1;
|
|
4678
|
+
if (!Number.isSafeInteger(samples) || samples < 1) throw new RangeError("dspyRlm.samples must be a positive safe integer");
|
|
4679
|
+
if (samples > 1 && binding.consensus === void 0) throw new AnalystExpressivenessError(`samples > 1 needs a consensus port; definition '${definition.id}' declares none`);
|
|
4258
4680
|
const pricing = config.pricing ?? pricingForModel$1(config.model);
|
|
4259
4681
|
const engine = createDspyRlmTraceEngine({
|
|
4260
4682
|
call: config.call,
|
|
@@ -4277,18 +4699,26 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4277
4699
|
...config.dspyRlm?.traceToolTimeoutMs === void 0 ? {} : { traceToolTimeoutMs: config.dspyRlm.traceToolTimeoutMs },
|
|
4278
4700
|
...config.dspyRlm?.runner ? { runner: config.dspyRlm.runner } : {}
|
|
4279
4701
|
});
|
|
4280
|
-
const
|
|
4281
|
-
const
|
|
4282
|
-
|
|
4702
|
+
const protocolSha256 = definition.protocolSha256;
|
|
4703
|
+
const traceDefinition = {
|
|
4704
|
+
id: binding.traceAnalystId,
|
|
4705
|
+
description: definition.description,
|
|
4706
|
+
area,
|
|
4707
|
+
version: definition.version,
|
|
4708
|
+
question: definition.question,
|
|
4709
|
+
instructions,
|
|
4710
|
+
toolGroup: projection.toolGroup,
|
|
4711
|
+
limits
|
|
4712
|
+
};
|
|
4283
4713
|
const { instructionsOverride: _rlmOnlyOverride, ...directConfig } = config;
|
|
4284
|
-
const abstentionFallbackRunner =
|
|
4714
|
+
const abstentionFallbackRunner = binding.abstentionFallback({
|
|
4285
4715
|
...directConfig,
|
|
4286
4716
|
costLedger
|
|
4287
4717
|
});
|
|
4288
4718
|
return {
|
|
4289
|
-
id:
|
|
4719
|
+
id: definition.id,
|
|
4290
4720
|
async analyze(input, context) {
|
|
4291
|
-
const trajectoryId =
|
|
4721
|
+
const trajectoryId = binding.subjectFromCaseId(context.caseId);
|
|
4292
4722
|
const tags = {
|
|
4293
4723
|
benchmarkCaseId: context.caseId,
|
|
4294
4724
|
benchmarkRepetition: String(context.repetition)
|
|
@@ -4296,7 +4726,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4296
4726
|
let usage;
|
|
4297
4727
|
let rawFindings = [];
|
|
4298
4728
|
try {
|
|
4299
|
-
if (!input.traceStore) throw new Error(
|
|
4729
|
+
if (!input.traceStore) throw new Error(`repl-variable analyst '${definition.id}' requires a trace store`);
|
|
4300
4730
|
if (samples > 1) {
|
|
4301
4731
|
const store = input.traceStore;
|
|
4302
4732
|
const caseUsageFilter = {
|
|
@@ -4310,14 +4740,14 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4310
4740
|
for (let sample = 0; sample < samples; sample += 1) {
|
|
4311
4741
|
let sampleUsage;
|
|
4312
4742
|
const completed = await runTraceAnalyst({
|
|
4313
|
-
definition,
|
|
4743
|
+
definition: traceDefinition,
|
|
4314
4744
|
engine,
|
|
4315
4745
|
store,
|
|
4316
4746
|
context: {
|
|
4317
4747
|
runId: context.caseId,
|
|
4318
4748
|
correlationId: `${context.caseId}:${context.repetition}:sample-${sample}`,
|
|
4319
4749
|
costLedger,
|
|
4320
|
-
costPhase:
|
|
4750
|
+
costPhase: binding.costPhase,
|
|
4321
4751
|
tags,
|
|
4322
4752
|
recordUsage: (receipt) => {
|
|
4323
4753
|
sampleUsage = receipt;
|
|
@@ -4328,8 +4758,8 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4328
4758
|
});
|
|
4329
4759
|
const producedAt = (/* @__PURE__ */ new Date()).toISOString();
|
|
4330
4760
|
const sampleFindings = completed.findings.map((finding) => makeFinding({
|
|
4331
|
-
analyst_id:
|
|
4332
|
-
area
|
|
4761
|
+
analyst_id: definition.id,
|
|
4762
|
+
area,
|
|
4333
4763
|
subject: finding.subject,
|
|
4334
4764
|
claim: finding.claim,
|
|
4335
4765
|
rationale: finding.rationale,
|
|
@@ -4338,20 +4768,18 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4338
4768
|
evidence_refs: evidenceRefsFromRawFinding(finding),
|
|
4339
4769
|
recommended_action: finding.recommended_action,
|
|
4340
4770
|
metadata: {
|
|
4341
|
-
|
|
4342
|
-
engine: "dspy-rlm",
|
|
4771
|
+
...binding.findingBaseMetadata,
|
|
4343
4772
|
model: config.model,
|
|
4344
4773
|
sample,
|
|
4345
|
-
...
|
|
4774
|
+
...binding.metadataFromSubject?.(finding.subject) ?? {}
|
|
4346
4775
|
},
|
|
4347
4776
|
produced_at: producedAt
|
|
4348
4777
|
}));
|
|
4349
4778
|
rawFindings = [...rawFindings, ...sampleFindings];
|
|
4350
|
-
const adapted = await
|
|
4351
|
-
|
|
4352
|
-
trajectoryId,
|
|
4779
|
+
const adapted = await binding.adapt({
|
|
4780
|
+
subject: trajectoryId,
|
|
4353
4781
|
findings: sampleFindings,
|
|
4354
|
-
analystId:
|
|
4782
|
+
analystId: definition.id,
|
|
4355
4783
|
store,
|
|
4356
4784
|
...context.signal ? { signal: context.signal } : {}
|
|
4357
4785
|
});
|
|
@@ -4366,18 +4794,17 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4366
4794
|
modelCalls: completed.modelCalls,
|
|
4367
4795
|
toolCalls: completed.toolCalls,
|
|
4368
4796
|
runtime: completed.runtime,
|
|
4369
|
-
|
|
4370
|
-
steps: assignments.map((assignment) => assignment.step),
|
|
4797
|
+
...binding.consensus.sampleRecord(assignments),
|
|
4371
4798
|
...adapted.diagnostics ? { blockDiagnostics: adapted.diagnostics } : {},
|
|
4372
4799
|
...sampleUsage ? { usage: sampleUsage } : {}
|
|
4373
4800
|
});
|
|
4374
4801
|
}
|
|
4375
|
-
const consensus =
|
|
4376
|
-
const expanded = await
|
|
4377
|
-
trajectoryId,
|
|
4802
|
+
const consensus = binding.consensus.vote(sampleAssignments);
|
|
4803
|
+
const expanded = await binding.consensus.expand({
|
|
4804
|
+
subject: trajectoryId,
|
|
4378
4805
|
blocks: consensus.blocks,
|
|
4379
4806
|
store,
|
|
4380
|
-
analystId:
|
|
4807
|
+
analystId: definition.id,
|
|
4381
4808
|
producedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
4382
4809
|
...context.signal ? { signal: context.signal } : {}
|
|
4383
4810
|
});
|
|
@@ -4388,8 +4815,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4388
4815
|
findings: fallback && !fallback.error ? fallback.findings : expanded.findings,
|
|
4389
4816
|
usage,
|
|
4390
4817
|
metadata: {
|
|
4391
|
-
|
|
4392
|
-
engine: "dspy-rlm",
|
|
4818
|
+
...binding.baseMetadata,
|
|
4393
4819
|
protocolSha256,
|
|
4394
4820
|
samples,
|
|
4395
4821
|
sampleRuns,
|
|
@@ -4406,14 +4832,14 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4406
4832
|
};
|
|
4407
4833
|
}
|
|
4408
4834
|
const completed = await runTraceAnalyst({
|
|
4409
|
-
definition,
|
|
4835
|
+
definition: traceDefinition,
|
|
4410
4836
|
engine,
|
|
4411
4837
|
store: input.traceStore,
|
|
4412
4838
|
context: {
|
|
4413
4839
|
runId: context.caseId,
|
|
4414
4840
|
correlationId: `${context.caseId}:${context.repetition}`,
|
|
4415
4841
|
costLedger,
|
|
4416
|
-
costPhase:
|
|
4842
|
+
costPhase: binding.costPhase,
|
|
4417
4843
|
tags,
|
|
4418
4844
|
recordUsage: (receipt) => {
|
|
4419
4845
|
usage = receipt;
|
|
@@ -4423,8 +4849,8 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4423
4849
|
});
|
|
4424
4850
|
const producedAt = (/* @__PURE__ */ new Date()).toISOString();
|
|
4425
4851
|
rawFindings = completed.findings.map((finding) => makeFinding({
|
|
4426
|
-
analyst_id:
|
|
4427
|
-
area
|
|
4852
|
+
analyst_id: definition.id,
|
|
4853
|
+
area,
|
|
4428
4854
|
subject: finding.subject,
|
|
4429
4855
|
claim: finding.claim,
|
|
4430
4856
|
rationale: finding.rationale,
|
|
@@ -4433,18 +4859,16 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4433
4859
|
evidence_refs: evidenceRefsFromRawFinding(finding),
|
|
4434
4860
|
recommended_action: finding.recommended_action,
|
|
4435
4861
|
metadata: {
|
|
4436
|
-
|
|
4437
|
-
engine: "dspy-rlm",
|
|
4862
|
+
...binding.findingBaseMetadata,
|
|
4438
4863
|
model: config.model,
|
|
4439
|
-
...
|
|
4864
|
+
...binding.metadataFromSubject?.(finding.subject) ?? {}
|
|
4440
4865
|
},
|
|
4441
4866
|
produced_at: producedAt
|
|
4442
4867
|
}));
|
|
4443
|
-
const adapted = await
|
|
4444
|
-
|
|
4445
|
-
trajectoryId,
|
|
4868
|
+
const adapted = await binding.adapt({
|
|
4869
|
+
subject: trajectoryId,
|
|
4446
4870
|
findings: rawFindings,
|
|
4447
|
-
analystId:
|
|
4871
|
+
analystId: definition.id,
|
|
4448
4872
|
store: input.traceStore,
|
|
4449
4873
|
...context.signal ? { signal: context.signal } : {}
|
|
4450
4874
|
});
|
|
@@ -4463,8 +4887,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4463
4887
|
findings: fallback && !fallback.error ? fallback.findings : adapted.findings,
|
|
4464
4888
|
usage,
|
|
4465
4889
|
metadata: {
|
|
4466
|
-
|
|
4467
|
-
engine: "dspy-rlm",
|
|
4890
|
+
...binding.baseMetadata,
|
|
4468
4891
|
protocolSha256,
|
|
4469
4892
|
...adapted.diagnostics ? { blockDiagnostics: adapted.diagnostics } : {},
|
|
4470
4893
|
answer: completed.answer,
|
|
@@ -4487,8 +4910,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4487
4910
|
usage,
|
|
4488
4911
|
error: publicBenchmarkError(error, []),
|
|
4489
4912
|
metadata: {
|
|
4490
|
-
|
|
4491
|
-
engine: "dspy-rlm",
|
|
4913
|
+
...binding.baseMetadata,
|
|
4492
4914
|
...samples > 1 ? { samples } : {},
|
|
4493
4915
|
rawFindings
|
|
4494
4916
|
}
|
|
@@ -4516,18 +4938,6 @@ function sampleBlockRecords(assignments) {
|
|
|
4516
4938
|
acceptedSteps
|
|
4517
4939
|
}));
|
|
4518
4940
|
}
|
|
4519
|
-
function publicBenchmarkDefinition(dataset, limits, instructions) {
|
|
4520
|
-
return {
|
|
4521
|
-
id: dataset === "agentrx" ? "agentrx-dspy-rlm" : "codetracebench-dspy-rlm",
|
|
4522
|
-
description: dataset === "agentrx" ? "Localizes the first unrecoverable root-cause step." : "Localizes every incorrect state-changing assistant step.",
|
|
4523
|
-
area: dataset === "agentrx" ? "root-cause" : "incorrect",
|
|
4524
|
-
version: "1.0.0",
|
|
4525
|
-
question: dataset === "agentrx" ? "What is the first unrecoverable root cause in this failed trajectory?" : "Which assistant steps are incorrect under the CodeTraceBench definition?",
|
|
4526
|
-
instructions,
|
|
4527
|
-
toolGroup: "singleTrace",
|
|
4528
|
-
limits
|
|
4529
|
-
};
|
|
4530
|
-
}
|
|
4531
4941
|
function pricingForModel$1(model) {
|
|
4532
4942
|
const pricing = resolveModelPricing(model);
|
|
4533
4943
|
if (!pricing) throw new Error(`no pricing is configured for '${model}'; provide PublicAnalystBenchmarkModelConfig.pricing`);
|
|
@@ -4891,455 +5301,31 @@ function nodeHttpPrimeBridgeTransport() {
|
|
|
4891
5301
|
};
|
|
4892
5302
|
}
|
|
4893
5303
|
//#endregion
|
|
4894
|
-
//#region src/analyst/prime-protocol.ts
|
|
4895
|
-
function buildPrimePrompt(spec) {
|
|
4896
|
-
return [
|
|
4897
|
-
`QUESTION: ${spec.question}`,
|
|
4898
|
-
"",
|
|
4899
|
-
...spec.taskDefinition === void 0 ? [] : [
|
|
4900
|
-
"TASK DEFINITION:",
|
|
4901
|
-
spec.taskDefinition,
|
|
4902
|
-
""
|
|
4903
|
-
],
|
|
4904
|
-
...spec.contractLines,
|
|
4905
|
-
"",
|
|
4906
|
-
spec.trajectoryHeader,
|
|
4907
|
-
spec.renderedTrajectory,
|
|
4908
|
-
...spec.trailer === void 0 ? [] : ["", spec.trailer]
|
|
4909
|
-
].join("\n");
|
|
4910
|
-
}
|
|
4911
|
-
/** Carries the malformed reply and the contract — never the trajectory. */
|
|
4912
|
-
function buildPrimeRepairPrompt(spec) {
|
|
4913
|
-
return [
|
|
4914
|
-
"Your previous reply to a trace-analysis task was structurally malformed and could not be parsed",
|
|
4915
|
-
`(${spec.defect}). Below is your previous reply verbatim. Re-emit ONLY the corrected JSON — one`,
|
|
4916
|
-
"fenced ```json block, no other text, no tools. The JSON object has exactly two fields:",
|
|
4917
|
-
...spec.repairContractLines,
|
|
4918
|
-
"",
|
|
4919
|
-
"PREVIOUS REPLY:",
|
|
4920
|
-
spec.previousReply
|
|
4921
|
-
].join("\n");
|
|
4922
|
-
}
|
|
4923
|
-
/**
|
|
4924
|
-
* Recover the reply's JSON object.
|
|
4925
|
-
*
|
|
4926
|
-
* Distinct from `extractJsonPayload` in ../llm-client, which serves a response
|
|
4927
|
-
* that DECLARES a JSON root and therefore must not scan onward. A prime reply
|
|
4928
|
-
* is prose plus a fenced block, and when the model emits several fences the
|
|
4929
|
-
* last one is its answer — so fences are scanned in reverse, and only then is a
|
|
4930
|
-
* brace-to-brace slice tried.
|
|
4931
|
-
*/
|
|
4932
|
-
function extractPrimeJsonObject(text) {
|
|
4933
|
-
const direct = parsePrimeJsonObject(text);
|
|
4934
|
-
if (direct) return direct;
|
|
4935
|
-
const fenced = [...text.matchAll(/```(?:json)?\s*\n?([\s\S]*?)```/g)];
|
|
4936
|
-
for (let index = fenced.length - 1; index >= 0; index -= 1) {
|
|
4937
|
-
const candidate = parsePrimeJsonObject(fenced[index][1]);
|
|
4938
|
-
if (candidate) return candidate;
|
|
4939
|
-
}
|
|
4940
|
-
const start = text.indexOf("{");
|
|
4941
|
-
const end = text.lastIndexOf("}");
|
|
4942
|
-
if (start >= 0 && end > start) {
|
|
4943
|
-
const candidate = parsePrimeJsonObject(text.slice(start, end + 1));
|
|
4944
|
-
if (candidate) return candidate;
|
|
4945
|
-
}
|
|
4946
|
-
return null;
|
|
4947
|
-
}
|
|
4948
|
-
/** Why the reply cannot be read as a prime answer, or null when it can. */
|
|
4949
|
-
function primeReplyDefect(parsed, rowsField) {
|
|
4950
|
-
if (parsed === null) return "no parseable JSON object";
|
|
4951
|
-
if (!Array.isArray(parsed[rowsField])) return `JSON has no "${rowsField}" array`;
|
|
4952
|
-
return null;
|
|
4953
|
-
}
|
|
4954
|
-
function parsePrimeJsonObject(text) {
|
|
4955
|
-
try {
|
|
4956
|
-
const value = JSON.parse(text.trim());
|
|
4957
|
-
return typeof value === "object" && value !== null && !Array.isArray(value) ? value : null;
|
|
4958
|
-
} catch {
|
|
4959
|
-
return null;
|
|
4960
|
-
}
|
|
4961
|
-
}
|
|
4962
|
-
function emptyPrimeRawUsage() {
|
|
4963
|
-
return {
|
|
4964
|
-
calls: null,
|
|
4965
|
-
inputTokens: null,
|
|
4966
|
-
outputTokens: null,
|
|
4967
|
-
bridgeEstimated: false
|
|
4968
|
-
};
|
|
4969
|
-
}
|
|
4970
|
-
/** Read the bridge's OpenAI-shaped `usage` object. */
|
|
4971
|
-
function normalizePrimeUsage(raw) {
|
|
4972
|
-
if (typeof raw !== "object" || raw === null) return emptyPrimeRawUsage();
|
|
4973
|
-
const record = raw;
|
|
4974
|
-
return {
|
|
4975
|
-
calls: tokenCountOrNull(record.model_requests),
|
|
4976
|
-
inputTokens: tokenCountOrNull(record.prompt_tokens),
|
|
4977
|
-
outputTokens: tokenCountOrNull(record.completion_tokens),
|
|
4978
|
-
bridgeEstimated: record.estimated === true
|
|
4979
|
-
};
|
|
4980
|
-
}
|
|
4981
|
-
/**
|
|
4982
|
-
* Sum two turns. Each side poisons independently: two turns that both report
|
|
4983
|
-
* input and neither report output yield a real input total beside a null
|
|
4984
|
-
* output, because discarding a measured count is as wrong as inventing one.
|
|
4985
|
-
*/
|
|
4986
|
-
function mergePrimeRawUsage(a, b) {
|
|
4987
|
-
return {
|
|
4988
|
-
calls: sumOrNull(a.calls, b.calls),
|
|
4989
|
-
inputTokens: sumOrNull(a.inputTokens, b.inputTokens),
|
|
4990
|
-
outputTokens: sumOrNull(a.outputTokens, b.outputTokens),
|
|
4991
|
-
bridgeEstimated: a.bridgeEstimated || b.bridgeEstimated
|
|
4992
|
-
};
|
|
4993
|
-
}
|
|
4994
|
-
function sumOrNull(a, b) {
|
|
4995
|
-
return a !== null && b !== null ? a + b : null;
|
|
4996
|
-
}
|
|
4997
|
-
function tokenCountOrNull(value) {
|
|
4998
|
-
return typeof value === "number" && Number.isSafeInteger(value) && value >= 0 ? value : null;
|
|
4999
|
-
}
|
|
5000
|
-
/**
|
|
5001
|
-
* Bind raw prime usage to agent-eval's typed receipt.
|
|
5002
|
-
*
|
|
5003
|
-
* `RunTokenUsage.input` and `.output` are non-nullable, so a one-sided count
|
|
5004
|
-
* cannot round-trip through `tokens` without writing a zero nobody measured.
|
|
5005
|
-
* The complete-accounting field therefore stays null, the reported side is
|
|
5006
|
-
* carried verbatim in `partialTokens`, and its price becomes the receipt's
|
|
5007
|
-
* `knownCostUsd` lower bound.
|
|
5008
|
-
*
|
|
5009
|
-
* Only agent-eval calls this; consumers with no pricing table read
|
|
5010
|
-
* `PrimeRawUsage` directly.
|
|
5011
|
-
*/
|
|
5012
|
-
function analystUsageReceiptFromPrimeUsage(usage, pricing) {
|
|
5013
|
-
const { calls, inputTokens, outputTokens, bridgeEstimated } = usage;
|
|
5014
|
-
const estimatedTokens = bridgeEstimated ? { tokensEstimated: true } : {};
|
|
5015
|
-
if (inputTokens !== null && outputTokens !== null) return {
|
|
5016
|
-
calls,
|
|
5017
|
-
tokens: {
|
|
5018
|
-
input: inputTokens,
|
|
5019
|
-
output: outputTokens
|
|
5020
|
-
},
|
|
5021
|
-
cost: {
|
|
5022
|
-
kind: "estimated",
|
|
5023
|
-
usd: priceTokens(inputTokens, outputTokens, pricing)
|
|
5024
|
-
},
|
|
5025
|
-
...estimatedTokens
|
|
5026
|
-
};
|
|
5027
|
-
if (inputTokens === null && outputTokens === null) return {
|
|
5028
|
-
calls,
|
|
5029
|
-
tokens: null,
|
|
5030
|
-
cost: {
|
|
5031
|
-
kind: "uncaptured",
|
|
5032
|
-
usd: null
|
|
5033
|
-
},
|
|
5034
|
-
...estimatedTokens
|
|
5035
|
-
};
|
|
5036
|
-
return {
|
|
5037
|
-
calls,
|
|
5038
|
-
tokens: null,
|
|
5039
|
-
partialTokens: {
|
|
5040
|
-
input: inputTokens,
|
|
5041
|
-
output: outputTokens
|
|
5042
|
-
},
|
|
5043
|
-
cost: {
|
|
5044
|
-
kind: "uncaptured",
|
|
5045
|
-
usd: null
|
|
5046
|
-
},
|
|
5047
|
-
knownCostUsd: priceTokens(inputTokens ?? 0, outputTokens ?? 0, pricing),
|
|
5048
|
-
...estimatedTokens
|
|
5049
|
-
};
|
|
5050
|
-
}
|
|
5051
|
-
function priceTokens(input, output, pricing) {
|
|
5052
|
-
return (input * pricing.inputUsdPerMillion + output * pricing.outputUsdPerMillion) / 1e6;
|
|
5053
|
-
}
|
|
5054
|
-
/**
|
|
5055
|
-
* Run the protocol: one call, one bounded repair turn on a structurally
|
|
5056
|
-
* malformed reply, then decode. Zero valid rows from a well-formed reply is an
|
|
5057
|
-
* honest null, not a failure.
|
|
5058
|
-
*/
|
|
5059
|
-
async function runPrimeExchange(options) {
|
|
5060
|
-
const { contract } = options;
|
|
5061
|
-
const turns = [];
|
|
5062
|
-
const repair = {
|
|
5063
|
-
attempted: false,
|
|
5064
|
-
succeeded: null
|
|
5065
|
-
};
|
|
5066
|
-
const first = await callPrimeTurn(options, options.prompt);
|
|
5067
|
-
if (!first.ok) return {
|
|
5068
|
-
ok: false,
|
|
5069
|
-
failure: first.failure,
|
|
5070
|
-
usage: mergeTurns(turns),
|
|
5071
|
-
turns,
|
|
5072
|
-
repair
|
|
5073
|
-
};
|
|
5074
|
-
turns.push({
|
|
5075
|
-
turn: "first",
|
|
5076
|
-
usage: first.usage,
|
|
5077
|
-
rawUsage: first.rawUsage
|
|
5078
|
-
});
|
|
5079
|
-
let reply = first.content;
|
|
5080
|
-
let parsed = extractPrimeJsonObject(reply);
|
|
5081
|
-
let defect = primeReplyDefect(parsed, contract.rowsField);
|
|
5082
|
-
if (defect !== null && options.repair) {
|
|
5083
|
-
repair.attempted = true;
|
|
5084
|
-
const second = await callPrimeTurn(options, buildPrimeRepairPrompt({
|
|
5085
|
-
defect,
|
|
5086
|
-
previousReply: reply,
|
|
5087
|
-
repairContractLines: contract.repairContractLines
|
|
5088
|
-
}));
|
|
5089
|
-
if (!second.ok) return {
|
|
5090
|
-
ok: false,
|
|
5091
|
-
failure: second.failure,
|
|
5092
|
-
usage: mergeTurns(turns),
|
|
5093
|
-
turns,
|
|
5094
|
-
repair,
|
|
5095
|
-
reply
|
|
5096
|
-
};
|
|
5097
|
-
turns.push({
|
|
5098
|
-
turn: "repair",
|
|
5099
|
-
usage: second.usage,
|
|
5100
|
-
rawUsage: second.rawUsage
|
|
5101
|
-
});
|
|
5102
|
-
reply = second.content;
|
|
5103
|
-
parsed = extractPrimeJsonObject(reply);
|
|
5104
|
-
defect = primeReplyDefect(parsed, contract.rowsField);
|
|
5105
|
-
repair.succeeded = defect === null;
|
|
5106
|
-
}
|
|
5107
|
-
const usage = mergeTurns(turns);
|
|
5108
|
-
if (defect !== null) return {
|
|
5109
|
-
ok: false,
|
|
5110
|
-
failure: {
|
|
5111
|
-
kind: "malformed-reply",
|
|
5112
|
-
message: `${defect} in prime reply${repair.attempted ? " even after the bounded repair turn" : ""}`
|
|
5113
|
-
},
|
|
5114
|
-
usage,
|
|
5115
|
-
turns,
|
|
5116
|
-
repair,
|
|
5117
|
-
reply
|
|
5118
|
-
};
|
|
5119
|
-
const rawRows = parsed[contract.rowsField];
|
|
5120
|
-
const rows = [];
|
|
5121
|
-
const rejected = [];
|
|
5122
|
-
let overflow = 0;
|
|
5123
|
-
rawRows.forEach((row, index) => {
|
|
5124
|
-
const decoded = contract.decodeRow(row, index);
|
|
5125
|
-
if (!decoded.ok) {
|
|
5126
|
-
rejected.push({
|
|
5127
|
-
index,
|
|
5128
|
-
reason: decoded.reason
|
|
5129
|
-
});
|
|
5130
|
-
return;
|
|
5131
|
-
}
|
|
5132
|
-
if (contract.maxRows !== void 0 && rows.length >= contract.maxRows) {
|
|
5133
|
-
overflow += 1;
|
|
5134
|
-
return;
|
|
5135
|
-
}
|
|
5136
|
-
rows.push(decoded.row);
|
|
5137
|
-
});
|
|
5138
|
-
const answer = parsed.answer;
|
|
5139
|
-
return {
|
|
5140
|
-
ok: true,
|
|
5141
|
-
answer: typeof answer === "string" ? answer : null,
|
|
5142
|
-
rows,
|
|
5143
|
-
rejected,
|
|
5144
|
-
reportedRows: rawRows.length,
|
|
5145
|
-
overflow,
|
|
5146
|
-
usage,
|
|
5147
|
-
turns,
|
|
5148
|
-
repair,
|
|
5149
|
-
reply
|
|
5150
|
-
};
|
|
5151
|
-
}
|
|
5152
|
-
async function callPrimeTurn(options, content) {
|
|
5153
|
-
const { transport, url, model, timeoutMs, signal } = options;
|
|
5154
|
-
const controller = new AbortController();
|
|
5155
|
-
const forwardAbort = () => controller.abort(signal?.reason);
|
|
5156
|
-
if (signal?.aborted) controller.abort(signal.reason);
|
|
5157
|
-
else signal?.addEventListener("abort", forwardAbort, { once: true });
|
|
5158
|
-
const deadline = setTimeout(() => controller.abort(), timeoutMs);
|
|
5159
|
-
let result;
|
|
5160
|
-
try {
|
|
5161
|
-
result = await transport({
|
|
5162
|
-
url,
|
|
5163
|
-
body: {
|
|
5164
|
-
model,
|
|
5165
|
-
messages: [{
|
|
5166
|
-
role: "user",
|
|
5167
|
-
content
|
|
5168
|
-
}]
|
|
5169
|
-
},
|
|
5170
|
-
signal: controller.signal
|
|
5171
|
-
});
|
|
5172
|
-
} catch (error) {
|
|
5173
|
-
if (signal?.aborted) return {
|
|
5174
|
-
ok: false,
|
|
5175
|
-
failure: {
|
|
5176
|
-
kind: "aborted",
|
|
5177
|
-
message: "prime exchange cancelled by the caller",
|
|
5178
|
-
cause: error
|
|
5179
|
-
}
|
|
5180
|
-
};
|
|
5181
|
-
if (controller.signal.aborted) return {
|
|
5182
|
-
ok: false,
|
|
5183
|
-
failure: {
|
|
5184
|
-
kind: "deadline",
|
|
5185
|
-
message: `bridge call exceeded ${timeoutMs}ms`
|
|
5186
|
-
}
|
|
5187
|
-
};
|
|
5188
|
-
return {
|
|
5189
|
-
ok: false,
|
|
5190
|
-
failure: {
|
|
5191
|
-
kind: "transport",
|
|
5192
|
-
message: `bridge transport failure: ${error instanceof Error ? error.message : String(error)}`
|
|
5193
|
-
}
|
|
5194
|
-
};
|
|
5195
|
-
} finally {
|
|
5196
|
-
clearTimeout(deadline);
|
|
5197
|
-
signal?.removeEventListener("abort", forwardAbort);
|
|
5198
|
-
}
|
|
5199
|
-
if (result.status !== 200) {
|
|
5200
|
-
const bodySnippet = result.text.slice(0, 500);
|
|
5201
|
-
return {
|
|
5202
|
-
ok: false,
|
|
5203
|
-
failure: {
|
|
5204
|
-
kind: "http-status",
|
|
5205
|
-
message: `bridge HTTP ${result.status}: ${bodySnippet}`,
|
|
5206
|
-
status: result.status,
|
|
5207
|
-
bodySnippet
|
|
5208
|
-
}
|
|
5209
|
-
};
|
|
5210
|
-
}
|
|
5211
|
-
let response;
|
|
5212
|
-
try {
|
|
5213
|
-
response = JSON.parse(result.text);
|
|
5214
|
-
} catch {
|
|
5215
|
-
return {
|
|
5216
|
-
ok: false,
|
|
5217
|
-
failure: {
|
|
5218
|
-
kind: "unparseable-json",
|
|
5219
|
-
message: `bridge returned unparseable JSON (${result.text.length} bytes)`
|
|
5220
|
-
}
|
|
5221
|
-
};
|
|
5222
|
-
}
|
|
5223
|
-
const replyContent = primeReplyContent(response);
|
|
5224
|
-
if (replyContent === null) return {
|
|
5225
|
-
ok: false,
|
|
5226
|
-
failure: {
|
|
5227
|
-
kind: "no-content",
|
|
5228
|
-
message: "bridge reply carries no message content"
|
|
5229
|
-
}
|
|
5230
|
-
};
|
|
5231
|
-
const rawUsage = primeReplyUsage(response);
|
|
5232
|
-
return {
|
|
5233
|
-
ok: true,
|
|
5234
|
-
content: replyContent,
|
|
5235
|
-
usage: normalizePrimeUsage(rawUsage),
|
|
5236
|
-
rawUsage
|
|
5237
|
-
};
|
|
5238
|
-
}
|
|
5239
|
-
/**
|
|
5240
|
-
* Fold from the FIRST turn, never from an empty receipt: an all-null identity
|
|
5241
|
-
* would poison every side it merged with and erase counts the bridge reported.
|
|
5242
|
-
*/
|
|
5243
|
-
function mergeTurns(turns) {
|
|
5244
|
-
if (turns.length === 0) return emptyPrimeRawUsage();
|
|
5245
|
-
return turns.slice(1).reduce((total, turn) => mergePrimeRawUsage(total, turn.usage), turns[0].usage);
|
|
5246
|
-
}
|
|
5247
|
-
function primeReplyContent(response) {
|
|
5248
|
-
if (typeof response !== "object" || response === null) return null;
|
|
5249
|
-
const choices = response.choices;
|
|
5250
|
-
if (!Array.isArray(choices) || choices.length === 0) return null;
|
|
5251
|
-
const message = choices[0]?.message;
|
|
5252
|
-
if (typeof message !== "object" || message === null) return null;
|
|
5253
|
-
const content = message.content;
|
|
5254
|
-
return typeof content === "string" && content.length > 0 ? content : null;
|
|
5255
|
-
}
|
|
5256
|
-
function primeReplyUsage(response) {
|
|
5257
|
-
if (typeof response !== "object" || response === null) return null;
|
|
5258
|
-
return response.usage ?? null;
|
|
5259
|
-
}
|
|
5260
|
-
/**
|
|
5261
|
-
* Render, measure, fall back to the capped projection, re-measure, fail loud.
|
|
5262
|
-
*
|
|
5263
|
-
* Inline is the only delivery prime has, so an oversized trajectory is a
|
|
5264
|
-
* refusal rather than a silent truncation: dropping spans would understate the
|
|
5265
|
-
* trajectory and the analyst would answer a question about a different run.
|
|
5266
|
-
*/
|
|
5267
|
-
async function projectPrimeTrajectory(source, limits) {
|
|
5268
|
-
let fetch = "full";
|
|
5269
|
-
let items = await source.full();
|
|
5270
|
-
if (items === null) {
|
|
5271
|
-
fetch = "capped";
|
|
5272
|
-
items = await source.capped();
|
|
5273
|
-
}
|
|
5274
|
-
let rendered = JSON.stringify(items);
|
|
5275
|
-
if (rendered.length > limits.maxInlineChars && fetch === "full") {
|
|
5276
|
-
fetch = "capped";
|
|
5277
|
-
items = await source.capped();
|
|
5278
|
-
rendered = JSON.stringify(items);
|
|
5279
|
-
}
|
|
5280
|
-
if (rendered.length > limits.maxInlineChars) return {
|
|
5281
|
-
ok: false,
|
|
5282
|
-
reason: `trajectory renders to ${rendered.length} chars even at ${source.cappedDescription}; inline delivery impossible`,
|
|
5283
|
-
renderedChars: rendered.length
|
|
5284
|
-
};
|
|
5285
|
-
return {
|
|
5286
|
-
ok: true,
|
|
5287
|
-
items,
|
|
5288
|
-
rendered,
|
|
5289
|
-
delivery: {
|
|
5290
|
-
mode: "inline-json",
|
|
5291
|
-
fetch,
|
|
5292
|
-
renderedChars: rendered.length
|
|
5293
|
-
}
|
|
5294
|
-
};
|
|
5295
|
-
}
|
|
5296
|
-
/**
|
|
5297
|
-
* Digest of everything a consumer can send to the bridge under the prime
|
|
5298
|
-
* protocol, recorded per observation so a prime result names the exact contract
|
|
5299
|
-
* that produced it.
|
|
5300
|
-
*
|
|
5301
|
-
* Computed over the ACTUALLY composed contract, so two consumers that both
|
|
5302
|
-
* stamp `analyst_id: 'prime'` while asking materially different questions get
|
|
5303
|
-
* different digests by construction. That is what makes 'prime' a reproducible
|
|
5304
|
-
* claim rather than a label.
|
|
5305
|
-
*/
|
|
5306
|
-
function primeProtocolSha256(identity) {
|
|
5307
|
-
return createHash("sha256").update(JSON.stringify({
|
|
5308
|
-
kind: "prime-analyst-protocol",
|
|
5309
|
-
question: identity.question,
|
|
5310
|
-
taskPrompt: identity.taskDefinition ?? null,
|
|
5311
|
-
outputContract: identity.contractLines,
|
|
5312
|
-
repairContract: buildPrimeRepairPrompt({
|
|
5313
|
-
defect: "<defect>",
|
|
5314
|
-
previousReply: "<previous-reply>",
|
|
5315
|
-
repairContractLines: identity.repairContractLines
|
|
5316
|
-
}),
|
|
5317
|
-
limits: identity.limits
|
|
5318
|
-
})).digest("hex");
|
|
5319
|
-
}
|
|
5320
|
-
//#endregion
|
|
5321
5304
|
//#region src/analyst/benchmark-runner-prime.ts
|
|
5322
5305
|
/**
|
|
5323
5306
|
* Prime analyst arm: the RLM coding agent reached through an OpenAI-compatible
|
|
5324
5307
|
* cli-bridge solves the CodeTraceBench incorrect-step task as a one-shot trace
|
|
5325
5308
|
* analyst.
|
|
5326
5309
|
*
|
|
5327
|
-
* The
|
|
5328
|
-
*
|
|
5329
|
-
*
|
|
5330
|
-
*
|
|
5331
|
-
*
|
|
5310
|
+
* The arm is expressed as an `AnalystDefinition`
|
|
5311
|
+
* (`primeCodeTraceAnalystDefinition`): the question, task text, output
|
|
5312
|
+
* contract, inline projection budget, and repair-turn declaration are all
|
|
5313
|
+
* definition content, and `createPrimeBenchmarkRunner` is a thin shell that
|
|
5314
|
+
* builds the definition and runs it through the inline strategy below. The
|
|
5315
|
+
* same strategy is what `bindAnalyst` (./bind) dispatches to, so a compiled
|
|
5316
|
+
* definition and this entry point send byte-identical requests — the parity
|
|
5317
|
+
* suite asserts exactly that.
|
|
5332
5318
|
*
|
|
5333
|
-
* The protocol
|
|
5319
|
+
* The protocol machinery — prompt composition, the bounded repair turn, reply
|
|
5334
5320
|
* extraction, the projection ladder, usage normalization — lives in
|
|
5335
|
-
* `./prime-protocol`, which knows nothing about CodeTraceBench. This file
|
|
5336
|
-
* the benchmark's binding to it
|
|
5337
|
-
*
|
|
5321
|
+
* `./prime-protocol`, which knows nothing about CodeTraceBench. This file adds
|
|
5322
|
+
* the benchmark's binding to it (block row grammar, store-backed projection,
|
|
5323
|
+
* observation shape) plus the projection-generic inline execution strategy.
|
|
5338
5324
|
*
|
|
5339
5325
|
* Trajectory delivery is inline JSON in the prompt. The dspy typed path binds
|
|
5340
5326
|
* the viewTrace span projection as a REPL variable; prime has no REPL, so the
|
|
5341
5327
|
* same projection is serialized into the prompt. When the full projection is
|
|
5342
|
-
* oversized the
|
|
5328
|
+
* oversized the strategy falls back to chunked viewSpans over the same
|
|
5343
5329
|
* projection surface with a per-attribute byte cap, and fails loud if the
|
|
5344
5330
|
* result still exceeds the inline budget.
|
|
5345
5331
|
*
|
|
@@ -5441,56 +5427,142 @@ const PRIME_BLOCK_CONTRACT = {
|
|
|
5441
5427
|
}
|
|
5442
5428
|
};
|
|
5443
5429
|
/**
|
|
5444
|
-
* Digest of everything this
|
|
5430
|
+
* Digest of everything this arm can send to the bridge, recorded per
|
|
5445
5431
|
* observation so a prime result names the exact contract that produced it.
|
|
5446
5432
|
*/
|
|
5447
5433
|
function primeAnalystProtocolSha256() {
|
|
5448
5434
|
return primeProtocolSha256(PRIME_PROTOCOL_IDENTITY);
|
|
5449
5435
|
}
|
|
5450
|
-
/**
|
|
5436
|
+
/**
|
|
5437
|
+
* The prime arm as a declarative unit. CodeTraceBench-only: the question,
|
|
5438
|
+
* task text, and block grammar speak its incorrect-step definition.
|
|
5439
|
+
*/
|
|
5440
|
+
function primeCodeTraceAnalystDefinition(args) {
|
|
5441
|
+
return {
|
|
5442
|
+
id: PRIME_ANALYST_ID,
|
|
5443
|
+
description: "One-shot RLM over an OpenAI-compatible bridge answering the CodeTraceBench incorrect-step task.",
|
|
5444
|
+
version: "1.0.0",
|
|
5445
|
+
area: "incorrect",
|
|
5446
|
+
profile: {},
|
|
5447
|
+
question: PRIME_QUESTION,
|
|
5448
|
+
taskDefinition: CODE_TRACE_BENCH_ANALYST_PROMPT,
|
|
5449
|
+
projection: {
|
|
5450
|
+
mode: "inline",
|
|
5451
|
+
maxInlineChars: MAX_INLINE_TRAJECTORY_CHARS,
|
|
5452
|
+
cappedAttributeBytes: CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP
|
|
5453
|
+
},
|
|
5454
|
+
replyContract: PRIME_BLOCK_CONTRACT,
|
|
5455
|
+
contractLimits: {
|
|
5456
|
+
maxBlocks: 16,
|
|
5457
|
+
maxBlockSteps: 12
|
|
5458
|
+
},
|
|
5459
|
+
budget: { timeoutMs: args.timeoutMs },
|
|
5460
|
+
repair: { turns: args.repairTurns },
|
|
5461
|
+
protocolSha256: primeAnalystProtocolSha256(),
|
|
5462
|
+
binding: {
|
|
5463
|
+
kind: "inline",
|
|
5464
|
+
subjectFromCaseId: trajectoryIdFromCaseId,
|
|
5465
|
+
baseMetadata: {
|
|
5466
|
+
analysisMode: "prime-rlm",
|
|
5467
|
+
engine: "prime"
|
|
5468
|
+
},
|
|
5469
|
+
header(subject, spans) {
|
|
5470
|
+
const stepSpans = spans.filter((span) => /^step-\d+$/.test(String(span.span_id)));
|
|
5471
|
+
if (stepSpans.length === 0) throw new PrimeTraceProjectionError(`no step-<n> spans in trace '${subject}'`);
|
|
5472
|
+
return `TRAJECTORY (trace_id ${subject}; ${stepSpans.length} assistant step spans; full span projection as JSON):`;
|
|
5473
|
+
},
|
|
5474
|
+
trailer(_subject, spans) {
|
|
5475
|
+
const finalVerification = spans.filter(isFinalVerificationSpan);
|
|
5476
|
+
return finalVerification.length > 0 ? `FINAL VERIFICATION SPANS:\n${JSON.stringify(finalVerification)}` : "FINAL VERIFICATION: unavailable for this trajectory — trace backward from the latest failure evidence inside the trajectory itself.";
|
|
5477
|
+
},
|
|
5478
|
+
async expandRows({ subject, rows, store, analystId, signal }) {
|
|
5479
|
+
const expanded = await expandCodeTraceFailureBlocks({
|
|
5480
|
+
trajectoryId: subject,
|
|
5481
|
+
blocks: rows,
|
|
5482
|
+
store,
|
|
5483
|
+
analystId,
|
|
5484
|
+
...signal ? { signal } : {}
|
|
5485
|
+
});
|
|
5486
|
+
return {
|
|
5487
|
+
findings: expanded.findings,
|
|
5488
|
+
diagnostics: expanded.diagnostics
|
|
5489
|
+
};
|
|
5490
|
+
}
|
|
5491
|
+
}
|
|
5492
|
+
};
|
|
5493
|
+
}
|
|
5494
|
+
/** Thin shell: validate options, declare the definition, run the inline strategy. */
|
|
5451
5495
|
function createPrimeBenchmarkRunner(options) {
|
|
5452
|
-
const baseUrl = requiredString(options.baseUrl, "baseUrl").replace(/\/+$/, "");
|
|
5453
|
-
const model = requiredString(options.model, "model");
|
|
5454
5496
|
const timeoutMs = positiveSafeInteger(options.timeoutMs, "timeoutMs");
|
|
5455
5497
|
const repair = options.repair;
|
|
5456
5498
|
if (typeof repair !== "boolean") throw new TypeError("repair must be a boolean");
|
|
5457
|
-
|
|
5458
|
-
|
|
5499
|
+
return runInlineAnalystDefinition(primeCodeTraceAnalystDefinition({
|
|
5500
|
+
timeoutMs,
|
|
5501
|
+
repairTurns: repair ? 1 : 0
|
|
5502
|
+
}), {
|
|
5503
|
+
baseUrl: options.baseUrl,
|
|
5504
|
+
model: options.model,
|
|
5505
|
+
...options.transport ? { transport: options.transport } : {},
|
|
5506
|
+
...options.pricing ? { pricing: options.pricing } : {}
|
|
5507
|
+
});
|
|
5508
|
+
}
|
|
5509
|
+
/**
|
|
5510
|
+
* Compile an inline-projection definition into a runnable arm. Projection,
|
|
5511
|
+
* prompt composition, the bounded repair turn, and usage accounting are all
|
|
5512
|
+
* driven by the definition; nothing in this strategy names a benchmark.
|
|
5513
|
+
*/
|
|
5514
|
+
function runInlineAnalystDefinition(definition, transports) {
|
|
5515
|
+
const { projection, binding } = definition;
|
|
5516
|
+
if (projection.mode !== "inline" || binding.kind !== "inline") throw new AnalystExpressivenessError(`the inline strategy compiles only inline projections; definition '${definition.id}' declares projection '${projection.mode}' with a '${binding.kind}' binding`);
|
|
5517
|
+
if (definition.repair.turns > 1) throw new AnalystExpressivenessError(`the inline exchange grants at most one bounded repair turn; definition '${definition.id}' declares ${definition.repair.turns}`);
|
|
5518
|
+
const baseUrl = requiredString(transports.baseUrl, "baseUrl").replace(/\/+$/, "");
|
|
5519
|
+
const model = requiredString(transports.model, "model");
|
|
5520
|
+
const timeoutMs = positiveSafeInteger(definition.budget.timeoutMs, "timeoutMs");
|
|
5521
|
+
const repair = definition.repair.turns === 1;
|
|
5522
|
+
const pricing = transports.pricing ?? pricingForModel(model);
|
|
5523
|
+
const transport = transports.transport ?? nodeHttpPrimeBridgeTransport();
|
|
5459
5524
|
const url = `${baseUrl}/v1/chat/completions`;
|
|
5460
5525
|
return {
|
|
5461
|
-
id:
|
|
5526
|
+
id: definition.id,
|
|
5462
5527
|
async analyze(input, context) {
|
|
5463
|
-
const
|
|
5528
|
+
const subject = binding.subjectFromCaseId(context.caseId);
|
|
5464
5529
|
let usage;
|
|
5465
5530
|
let metadata = {
|
|
5466
|
-
|
|
5467
|
-
engine: "prime",
|
|
5531
|
+
...binding.baseMetadata,
|
|
5468
5532
|
bridgeUrl: baseUrl,
|
|
5469
5533
|
model,
|
|
5470
|
-
protocolSha256:
|
|
5534
|
+
protocolSha256: definition.protocolSha256
|
|
5471
5535
|
};
|
|
5472
5536
|
try {
|
|
5473
5537
|
const store = input.traceStore;
|
|
5474
|
-
if (!store) throw new Error(
|
|
5475
|
-
const
|
|
5476
|
-
|
|
5538
|
+
if (!store) throw new Error(`inline analyst '${definition.id}' requires a trace store`);
|
|
5539
|
+
const storeContext = context.signal ? { signal: context.signal } : void 0;
|
|
5540
|
+
const projected = await projectPrimeTrajectory(inlineProjectionSource(store, subject, projection.cappedAttributeBytes, storeContext), { maxInlineChars: projection.maxInlineChars });
|
|
5541
|
+
if (!projected.ok) throw new PrimeTraceProjectionError(projected.reason);
|
|
5477
5542
|
const delivery = {
|
|
5478
|
-
mode:
|
|
5479
|
-
fetch:
|
|
5480
|
-
perAttributeByteCap:
|
|
5481
|
-
renderedChars:
|
|
5543
|
+
mode: projected.delivery.mode,
|
|
5544
|
+
fetch: projected.delivery.fetch === "full" ? "view-trace" : "view-spans-chunked",
|
|
5545
|
+
perAttributeByteCap: projected.delivery.fetch === "full" ? null : projection.cappedAttributeBytes,
|
|
5546
|
+
renderedChars: projected.delivery.renderedChars
|
|
5482
5547
|
};
|
|
5483
5548
|
metadata = {
|
|
5484
5549
|
...metadata,
|
|
5485
5550
|
delivery
|
|
5486
5551
|
};
|
|
5487
|
-
const prompt =
|
|
5552
|
+
const prompt = buildPrimePrompt({
|
|
5553
|
+
question: definition.question,
|
|
5554
|
+
...definition.taskDefinition === void 0 ? {} : { taskDefinition: definition.taskDefinition },
|
|
5555
|
+
contractLines: definition.replyContract.contractLines,
|
|
5556
|
+
trajectoryHeader: binding.header(subject, projected.items),
|
|
5557
|
+
renderedTrajectory: projected.rendered,
|
|
5558
|
+
trailer: binding.trailer(subject, projected.items)
|
|
5559
|
+
});
|
|
5488
5560
|
metadata = {
|
|
5489
5561
|
...metadata,
|
|
5490
5562
|
promptChars: prompt.length
|
|
5491
5563
|
};
|
|
5492
5564
|
const outcome = await runPrimeExchange({
|
|
5493
|
-
contract:
|
|
5565
|
+
contract: definition.replyContract,
|
|
5494
5566
|
prompt,
|
|
5495
5567
|
transport,
|
|
5496
5568
|
url,
|
|
@@ -5518,11 +5590,11 @@ function createPrimeBenchmarkRunner(options) {
|
|
|
5518
5590
|
};
|
|
5519
5591
|
throw primeFailureError(outcome.failure);
|
|
5520
5592
|
}
|
|
5521
|
-
const expanded = await
|
|
5522
|
-
|
|
5523
|
-
|
|
5593
|
+
const expanded = await binding.expandRows({
|
|
5594
|
+
subject,
|
|
5595
|
+
rows: outcome.rows,
|
|
5524
5596
|
store,
|
|
5525
|
-
analystId:
|
|
5597
|
+
analystId: definition.id,
|
|
5526
5598
|
...context.signal ? { signal: context.signal } : {}
|
|
5527
5599
|
});
|
|
5528
5600
|
return {
|
|
@@ -5553,13 +5625,13 @@ function createPrimeBenchmarkRunner(options) {
|
|
|
5553
5625
|
* the full viewTrace projection, or the chunked viewSpans projection at a
|
|
5554
5626
|
* per-attribute byte cap.
|
|
5555
5627
|
*/
|
|
5556
|
-
function
|
|
5628
|
+
function inlineProjectionSource(store, trajectoryId, cappedAttributeBytes, context) {
|
|
5557
5629
|
return {
|
|
5558
5630
|
async full() {
|
|
5559
5631
|
return (await store.viewTrace({ trace_id: trajectoryId }, context)).spans ?? null;
|
|
5560
5632
|
},
|
|
5561
|
-
capped: () => projectSpansChunked(store, trajectoryId, context),
|
|
5562
|
-
cappedDescription: `per-attribute cap ${
|
|
5633
|
+
capped: () => projectSpansChunked(store, trajectoryId, cappedAttributeBytes, context),
|
|
5634
|
+
cappedDescription: `per-attribute cap ${cappedAttributeBytes}`
|
|
5563
5635
|
};
|
|
5564
5636
|
}
|
|
5565
5637
|
/**
|
|
@@ -5568,7 +5640,7 @@ function codeTraceProjectionSource(store, trajectoryId, context) {
|
|
|
5568
5640
|
* id must project or the case fails loud — a silently dropped span would
|
|
5569
5641
|
* understate the trajectory.
|
|
5570
5642
|
*/
|
|
5571
|
-
async function projectSpansChunked(store, trajectoryId, context) {
|
|
5643
|
+
async function projectSpansChunked(store, trajectoryId, cappedAttributeBytes, context) {
|
|
5572
5644
|
const enumeration = await store.viewTrace({
|
|
5573
5645
|
trace_id: trajectoryId,
|
|
5574
5646
|
per_attribute_byte_cap: SPAN_ID_ENUMERATION_ATTRIBUTE_BYTE_CAP
|
|
@@ -5587,26 +5659,13 @@ async function projectSpansChunked(store, trajectoryId, context) {
|
|
|
5587
5659
|
const result = await store.viewSpans({
|
|
5588
5660
|
trace_id: trajectoryId,
|
|
5589
5661
|
span_ids: chunk,
|
|
5590
|
-
per_attribute_byte_cap:
|
|
5662
|
+
per_attribute_byte_cap: cappedAttributeBytes
|
|
5591
5663
|
}, context);
|
|
5592
5664
|
if (result.missing_span_ids.length > 0 || result.omitted_span_ids.length > 0 || result.spans.length !== chunk.length) throw new PrimeTraceProjectionError(`viewSpans projected ${result.spans.length}/${chunk.length} spans for chunk at ${index} of '${trajectoryId}'`);
|
|
5593
5665
|
projected.push(...result.spans);
|
|
5594
5666
|
}
|
|
5595
5667
|
return projected;
|
|
5596
5668
|
}
|
|
5597
|
-
function buildCodeTracePrompt(trajectoryId, spans, renderedSpans) {
|
|
5598
|
-
const stepSpans = spans.filter((span) => /^step-\d+$/.test(String(span.span_id)));
|
|
5599
|
-
if (stepSpans.length === 0) throw new PrimeTraceProjectionError(`no step-<n> spans in trace '${trajectoryId}'`);
|
|
5600
|
-
const finalVerification = spans.filter(isFinalVerificationSpan);
|
|
5601
|
-
return buildPrimePrompt({
|
|
5602
|
-
question: PRIME_QUESTION,
|
|
5603
|
-
taskDefinition: CODE_TRACE_BENCH_ANALYST_PROMPT,
|
|
5604
|
-
contractLines: PRIME_OUTPUT_CONTRACT_LINES,
|
|
5605
|
-
trajectoryHeader: `TRAJECTORY (trace_id ${trajectoryId}; ${stepSpans.length} assistant step spans; full span projection as JSON):`,
|
|
5606
|
-
renderedTrajectory: renderedSpans,
|
|
5607
|
-
trailer: finalVerification.length > 0 ? `FINAL VERIFICATION SPANS:\n${JSON.stringify(finalVerification)}` : "FINAL VERIFICATION: unavailable for this trajectory — trace backward from the latest failure evidence inside the trajectory itself."
|
|
5608
|
-
});
|
|
5609
|
-
}
|
|
5610
5669
|
/** Map the protocol's terminal reason onto this benchmark's typed error classes. */
|
|
5611
5670
|
function primeFailureError(failure) {
|
|
5612
5671
|
switch (failure.kind) {
|
|
@@ -6351,6 +6410,6 @@ function shellQuote(value) {
|
|
|
6351
6410
|
return /^[a-zA-Z0-9_./:@%+=,-]+$/.test(value) ? value : `'${value.replaceAll("'", "'\\''")}'`;
|
|
6352
6411
|
}
|
|
6353
6412
|
//#endregion
|
|
6354
|
-
export { ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as $, summarizeCodeTraceCalibration as A, publicBenchmarkSystemPrompt as B,
|
|
6413
|
+
export { ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as $, summarizeCodeTraceCalibration as A, publicBenchmarkSystemPrompt as B, analystDefinitionAsymmetries as C, expandCodeTraceFailureBlocks as D, emptyPublicBenchmarkRunner as E, CODE_TRACE_BENCH_ANALYST_PROMPT as F, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as G, appendVerificationArtifactsToOtlp as H, MAX_INCORRECT_BLOCKS as I, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as J, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as K, MAX_INCORRECT_BLOCK_STEPS as L, analystInstructionsOverrideFromText as M, effectiveAnalystProtocolSha256 as N, readAnalystBenchmarkArtifact as O, readAnalystInstructionsOverride as P, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as Q, publicBenchmarkProtocolSha256 as R, AnalystExpressivenessError as S, adaptPublicBenchmarkFindings as T, loadCodeTraceVerificationArtifacts as U, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as V, parseVerificationOutcome as W, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as X, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as Y, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as Z, runReplVariableAnalystDefinition as _, primeAnalystProtocolSha256 as a, ANALYST_BENCHMARK_OBSERVATIONS_FILE as at, runChunkedAnalystDefinition as b, nodeHttpPrimeBridgeTransport as c, summarizeAgentRxCalibration as ct, publicBenchmarkDistributions as d, agentRxBenchmarkCase as dt, analystBenchmarkDependencyLockDigest as et, publicBenchmarkSelectionReport as f, agentRxPredictionsToFindings as ft, rlmEngineLimits as g, publicRlmAnalystDefinition as h, normalizeBenchmarkLabel as ht, createPrimeBenchmarkRunner as i, ANALYST_BENCHMARK_MANIFEST_FILE as it, compareAnalystRunners as j, renderCodeTraceCalibrationMarkdown as k, loadPublicBenchmarkRows as l, codeTraceBenchCase as lt, createPublicBenchmarkRlmRunner as m, roundAgentRxStep as mt, runAnalystBenchmarkCommand as n, ANALYST_BENCHMARK_COST_LEDGER_FILE as nt, primeCodeTraceAnalystDefinition as o, AGENT_RX_UPSTREAM_REVISION as ot, selectPublicBenchmarkRows as p, normalizeAgentRxCategory as pt, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as q, renderAnalystBenchmarkMarkdown as r, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as rt, runInlineAnalystDefinition as s, renderAgentRxCalibrationMarkdown as st, ANALYST_BENCHMARK_HELP as t, analystBenchmarkImplementationDigest as tt, preparePublicAnalystBenchmark as u, codeTracerPredictionsToFindings as ut, createPublicBenchmarkDirectRunner as v, analystDefinitionProtocolSha256 as w, decodeReplyRows as x, publicDirectAnalystDefinition as y, publicBenchmarkRlmInstructions as z };
|
|
6355
6414
|
|
|
6356
|
-
//# sourceMappingURL=benchmark-command-
|
|
6415
|
+
//# sourceMappingURL=benchmark-command-BCafwNrf.js.map
|