@tangle-network/agent-eval 0.144.6 → 0.144.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -0
- package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +470 -88
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BKENp2s5.js} +662 -605
- package/dist/benchmark-command-BKENp2s5.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BEOkuvIg.js → benchmarks-BlPmjd88.js} +4 -4
- package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-BlPmjd88.js.map} +1 -1
- package/dist/campaign/index.d.ts +7 -5
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-CXsdyym7.js → campaign--HVSuvV0.js} +17 -301
- package/dist/campaign--HVSuvV0.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
- package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
- package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -8
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
- package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
- package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +214 -404
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +233 -649
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
- package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
- package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
- package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +2 -2
- package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
- package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +17 -7
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
- package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
- package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
- package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CkaI2ly4.js} +19 -654
- package/dist/skillopt-optimization-method-CkaI2ly4.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
- package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
- package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +2 -2
- package/dist/wire/index.js +1 -1
- package/docs/campaign-proposers.md +5 -0
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +19 -2
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
- package/dist/benchmark-command-CQd78YHt.js.map +0 -1
- package/dist/campaign-CXsdyym7.js.map +0 -1
- package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
- package/dist/index-4XwggC10.d.ts.map +0 -1
- package/dist/index-BIL5vxxt.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-Dv5BiKLE.js.map +0 -1
- package/dist/replay-Krvb114g.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
- package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
- package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
- package/dist/types-XMVEdrE_.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
|
@@ -1,16 +1,17 @@
|
|
|
1
1
|
import { c as ValidationError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
|
|
2
2
|
import { s as resolveModelPricing } from "./metrics-C9YY1OcL.js";
|
|
3
3
|
import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-DMFxsLKr.js";
|
|
4
|
-
import { c as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-
|
|
5
|
-
import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-BHIgPmzS.js";
|
|
4
|
+
import { c as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-DzvMUsS_.js";
|
|
5
|
+
import { D as RawAnalystFindingSchema, O as evidenceRefsFromRawFinding, T as RAW_FINDING_SCHEMA_PROMPT, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-BHIgPmzS.js";
|
|
6
6
|
import { o as makeFinding, r as usageReceiptFromCostLedger } from "./usage-receipt-EVI8B8Xu.js";
|
|
7
7
|
import { i as hashCanonical, r as canonicalString } from "./canonical-D011XM8r.js";
|
|
8
|
-
import {
|
|
9
|
-
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-
|
|
8
|
+
import { D as resolveExternalOptimizerProcessLimits, f as runWithCleanup, n as createRunCostLedger, p as startExternalOptimizerModelProxy, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-BMQEv1wG.js";
|
|
9
|
+
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-BiN49gK6.js";
|
|
10
10
|
import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-DXZIqu17.js";
|
|
11
11
|
import { E as pairedBootstrap } from "./statistics-ByxzSiOM.js";
|
|
12
12
|
import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-CKtTpRhv.js";
|
|
13
13
|
import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-B181aMF9.js";
|
|
14
|
+
import { c as primeProtocolSha256, d as runPrimeExchange, f as assertEqualDeclarativeTerms, n as buildPrimePrompt, t as analystUsageReceiptFromPrimeUsage, u as projectPrimeTrajectory } from "./prime-protocol-BfSalTfR.js";
|
|
14
15
|
import { z } from "zod";
|
|
15
16
|
import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
|
|
16
17
|
import * as nodePath from "node:path";
|
|
@@ -1207,7 +1208,7 @@ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
|
|
|
1207
1208
|
"package.json",
|
|
1208
1209
|
"pnpm-lock.yaml"
|
|
1209
1210
|
]);
|
|
1210
|
-
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
1211
|
+
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "788e0d4e228836e84b4cf31492c9a0f7c884efcce088623988f9c378c768ef7c";
|
|
1211
1212
|
/** The published benchmark evidence was produced at this package version, by
|
|
1212
1213
|
* the retired one-shot direct runner, before trace analysts moved to the
|
|
1213
1214
|
* recursive DSPy RLM engine. Both evidence digests below are historical facts
|
|
@@ -1252,8 +1253,10 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1252
1253
|
"src/analyst/benchmark-verification-artifacts.ts",
|
|
1253
1254
|
"src/analyst/benchmark-verification-outcome.ts",
|
|
1254
1255
|
"src/analyst/benchmark.ts",
|
|
1256
|
+
"src/analyst/definition.ts",
|
|
1255
1257
|
"src/analyst/dspy-rlm-engine.ts",
|
|
1256
1258
|
"src/analyst/engine.ts",
|
|
1259
|
+
"src/analyst/equal-terms.ts",
|
|
1257
1260
|
"src/analyst/exact-types.ts",
|
|
1258
1261
|
"src/analyst/finding-signature.ts",
|
|
1259
1262
|
"src/analyst/finding-subject.ts",
|
|
@@ -1261,6 +1264,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1261
1264
|
"src/analyst/parse-tolerant.ts",
|
|
1262
1265
|
"src/analyst/prime-bridge-transport.ts",
|
|
1263
1266
|
"src/analyst/prime-protocol.ts",
|
|
1267
|
+
"src/analyst/reply-contract.ts",
|
|
1264
1268
|
"src/analyst/tool-groups.ts",
|
|
1265
1269
|
"src/analyst/trace-tool-callback.ts",
|
|
1266
1270
|
"src/analyst/types.ts",
|
|
@@ -1311,7 +1315,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1311
1315
|
"src/trace/raw-provider-sink.ts",
|
|
1312
1316
|
"src/verdict-cache.ts"
|
|
1313
1317
|
]);
|
|
1314
|
-
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
1318
|
+
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "cc354effd79c8dfc8669c230706c67c63a06e3aff013f540806361263bae8860";
|
|
1315
1319
|
function analystBenchmarkImplementationDigest() {
|
|
1316
1320
|
return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256;
|
|
1317
1321
|
}
|
|
@@ -2238,19 +2242,24 @@ Cite the block's first step and its last step as trace://<URL-encoded-trace-id>/
|
|
|
2238
2242
|
Give the rationale as the concrete downstream evidence visible at the consequence step.
|
|
2239
2243
|
Submit as soon as every candidate failure block has a supported verdict.
|
|
2240
2244
|
Return no finding for a clean trajectory.`;
|
|
2241
|
-
/**
|
|
2242
|
-
|
|
2243
|
-
const fieldContract = dataset === "agentrx" ? AGENT_RX_JSON_CONTRACT : CODE_TRACE_JSON_CONTRACT;
|
|
2244
|
-
return `${publicBenchmarkTaskPrompt(dataset)}
|
|
2245
|
-
|
|
2246
|
-
${fieldContract}
|
|
2247
|
-
|
|
2248
|
-
Return exactly one JSON object with:
|
|
2245
|
+
/** Reply-envelope contract shared by both one-shot datasets. */
|
|
2246
|
+
const PUBLIC_BENCHMARK_ENVELOPE_CONTRACT = `Return exactly one JSON object with:
|
|
2249
2247
|
- "report": a concise evidence-based explanation, at most 4000 characters
|
|
2250
2248
|
- "findings": the strict finding array
|
|
2251
2249
|
Use an empty findings array when the trace does not support a finding.
|
|
2252
2250
|
Do not return a bare array, markdown, trace URIs, copied excerpts, or fields not listed above.
|
|
2253
2251
|
The runner constructs exact trace URIs and action previews from each selected step.`;
|
|
2252
|
+
/** Per-dataset field grammar for the one-shot JSON reply. */
|
|
2253
|
+
function publicBenchmarkFieldContract(dataset) {
|
|
2254
|
+
return dataset === "agentrx" ? AGENT_RX_JSON_CONTRACT : CODE_TRACE_JSON_CONTRACT;
|
|
2255
|
+
}
|
|
2256
|
+
/** One-shot JSON transport prompt for the direct runner. */
|
|
2257
|
+
function publicBenchmarkSystemPrompt(dataset) {
|
|
2258
|
+
return [
|
|
2259
|
+
publicBenchmarkTaskPrompt(dataset),
|
|
2260
|
+
publicBenchmarkFieldContract(dataset),
|
|
2261
|
+
PUBLIC_BENCHMARK_ENVELOPE_CONTRACT
|
|
2262
|
+
].join("\n\n");
|
|
2254
2263
|
}
|
|
2255
2264
|
/** Tool-loop prompt for the recursive runner. Same task, subject-encoded block. */
|
|
2256
2265
|
function publicBenchmarkRlmInstructions(dataset) {
|
|
@@ -2258,6 +2267,7 @@ function publicBenchmarkRlmInstructions(dataset) {
|
|
|
2258
2267
|
return `${publicBenchmarkTaskPrompt(dataset)}
|
|
2259
2268
|
${outputContract}`;
|
|
2260
2269
|
}
|
|
2270
|
+
/** Task text shared by every runner shape on one dataset. */
|
|
2261
2271
|
function publicBenchmarkTaskPrompt(dataset) {
|
|
2262
2272
|
return dataset === "agentrx" ? AGENT_RX_PROMPT : CODE_TRACE_BENCH_ANALYST_PROMPT;
|
|
2263
2273
|
}
|
|
@@ -3594,61 +3604,300 @@ function fileContext() {
|
|
|
3594
3604
|
};
|
|
3595
3605
|
}
|
|
3596
3606
|
//#endregion
|
|
3607
|
+
//#region src/analyst/definition.ts
|
|
3608
|
+
/**
|
|
3609
|
+
* AnalystDefinition — the declarative unit behind an analyst arm.
|
|
3610
|
+
*
|
|
3611
|
+
* An arm is one way of EXECUTING an analysis question: a one-shot JSON call, a
|
|
3612
|
+
* bridge-reached RLM, a recursive engine with trace tools. What the arm SAYS —
|
|
3613
|
+
* the question, the task text, the reply grammar, how evidence reaches the
|
|
3614
|
+
* model, the repair-turn and budget terms — is protocol, not execution, so it
|
|
3615
|
+
* lives here as one inspectable value. `bindAnalyst` (./bind) compiles a
|
|
3616
|
+
* definition plus a transport binding into a runnable arm, and the parity
|
|
3617
|
+
* suite holds the compiled arm to the byte against the arm's entry point, so a
|
|
3618
|
+
* definition cannot drift from what its arm actually sends.
|
|
3619
|
+
*
|
|
3620
|
+
* Three rules carried over from the repair-arm comparison contract
|
|
3621
|
+
* (trace-repair's `repairArmAsymmetries`), made structural here:
|
|
3622
|
+
*
|
|
3623
|
+
* one contract the reply grammar is a `ReplyContract` value on the
|
|
3624
|
+
* definition, never prose inside a runner body.
|
|
3625
|
+
* one repair turn `analystDefinitionAsymmetries` refuses a set whose
|
|
3626
|
+
* definitions declare unequal repair turns, because a second
|
|
3627
|
+
* attempt is a second sample the other arms never got.
|
|
3628
|
+
* declared difference what arms MAY differ in — the evidence projection, the
|
|
3629
|
+
* reasoning effort, the budget — is declared per definition
|
|
3630
|
+
* and rendered beside the comparison instead of being
|
|
3631
|
+
* inferred from two runners' source.
|
|
3632
|
+
*/
|
|
3633
|
+
/**
|
|
3634
|
+
* Thrown at bind time when a definition asks for something no strategy can
|
|
3635
|
+
* compile — an unknown projection × transport pair, a repair-turn count the
|
|
3636
|
+
* exchange machinery cannot grant, a reasoning effort the arm cannot map. The
|
|
3637
|
+
* message names the construct so an expressiveness gap is a loud, attributable
|
|
3638
|
+
* failure instead of a silently narrowed protocol.
|
|
3639
|
+
*/
|
|
3640
|
+
var AnalystExpressivenessError = class extends Error {};
|
|
3641
|
+
/**
|
|
3642
|
+
* Digest of everything a definition can send to its model. An inline
|
|
3643
|
+
* definition hashes under the historical prime-protocol domain, so its digest
|
|
3644
|
+
* equals the digest its bespoke arm always recorded; other projections hash
|
|
3645
|
+
* under the definition domain.
|
|
3646
|
+
*/
|
|
3647
|
+
function analystDefinitionProtocolSha256(definition) {
|
|
3648
|
+
const { projection, replyContract } = definition;
|
|
3649
|
+
if (projection.mode === "inline") return primeProtocolSha256({
|
|
3650
|
+
question: definition.question,
|
|
3651
|
+
...definition.taskDefinition === void 0 ? {} : { taskDefinition: definition.taskDefinition },
|
|
3652
|
+
contractLines: replyContract.contractLines,
|
|
3653
|
+
repairContractLines: replyContract.repairContractLines,
|
|
3654
|
+
limits: {
|
|
3655
|
+
...definition.contractLimits,
|
|
3656
|
+
maxInlineTrajectoryChars: projection.maxInlineChars,
|
|
3657
|
+
chunkedProjectionAttributeByteCap: projection.cappedAttributeBytes
|
|
3658
|
+
}
|
|
3659
|
+
});
|
|
3660
|
+
return createHash("sha256").update(JSON.stringify({
|
|
3661
|
+
kind: "analyst-definition-protocol",
|
|
3662
|
+
mode: projection.mode,
|
|
3663
|
+
question: definition.question,
|
|
3664
|
+
taskDefinition: definition.taskDefinition ?? null,
|
|
3665
|
+
contractLines: replyContract.contractLines,
|
|
3666
|
+
repairContractLines: replyContract.repairContractLines,
|
|
3667
|
+
limits: definition.contractLimits,
|
|
3668
|
+
projection: projection.mode === "chunked" ? { attributeByteCaps: projection.attributeByteCaps } : { toolGroup: projection.toolGroup }
|
|
3669
|
+
})).digest("hex");
|
|
3670
|
+
}
|
|
3671
|
+
/**
|
|
3672
|
+
* Refuse a set of definitions that cannot be compared on equal terms, and
|
|
3673
|
+
* render what still differs between the ones that can. The hard rule is the
|
|
3674
|
+
* repair turn: a malformed reply must earn the same number of retries in every
|
|
3675
|
+
* arm, because a retry is a second sample. Projection, reasoning effort, and
|
|
3676
|
+
* budget differences are declared and reported, never hidden.
|
|
3677
|
+
*/
|
|
3678
|
+
function analystDefinitionAsymmetries(definitions) {
|
|
3679
|
+
const { ids, repairTurns } = assertEqualDeclarativeTerms("analyst definition", definitions.map((definition) => ({
|
|
3680
|
+
id: definition.id,
|
|
3681
|
+
repairTurns: definition.repair.turns
|
|
3682
|
+
})));
|
|
3683
|
+
const firstMode = definitions[0].projection.mode;
|
|
3684
|
+
return {
|
|
3685
|
+
ids,
|
|
3686
|
+
repairTurns,
|
|
3687
|
+
sharedProjectionMode: definitions.every((definition) => definition.projection.mode === firstMode) ? firstMode : null,
|
|
3688
|
+
asymmetries: definitions.map((definition) => ({
|
|
3689
|
+
id: definition.id,
|
|
3690
|
+
projectionMode: definition.projection.mode,
|
|
3691
|
+
reasoningEffort: definition.profile.model?.reasoningEffort ?? null,
|
|
3692
|
+
timeoutMs: definition.budget.timeoutMs,
|
|
3693
|
+
maxCostUsd: definition.budget.maxCostUsd ?? null,
|
|
3694
|
+
maxOutputTokens: definition.budget.maxOutputTokens ?? null,
|
|
3695
|
+
protocolSha256: definition.protocolSha256,
|
|
3696
|
+
definitionSha256: analystDefinitionProtocolSha256(definition)
|
|
3697
|
+
}))
|
|
3698
|
+
};
|
|
3699
|
+
}
|
|
3700
|
+
//#endregion
|
|
3701
|
+
//#region src/analyst/reply-contract.ts
|
|
3702
|
+
/**
|
|
3703
|
+
* The reply grammar an analyst arm holds a model to, independent of transport.
|
|
3704
|
+
*
|
|
3705
|
+
* One contract serves every arm shape: the inline bridge protocol
|
|
3706
|
+
* (`runPrimeExchange`) reads the base fields, and one-shot JSON arms
|
|
3707
|
+
* additionally use the strict-envelope and all-rejected knobs. `PrimeReplyContract`
|
|
3708
|
+
* in ./prime-protocol is a type alias of this contract, so a consumer written
|
|
3709
|
+
* against the prime protocol names the same grammar object.
|
|
3710
|
+
*/
|
|
3711
|
+
/**
|
|
3712
|
+
* Decode a parsed reply value under a contract: strict envelope when declared,
|
|
3713
|
+
* then per-row decoding, then the all-rejected policy. Shape before count: a
|
|
3714
|
+
* malformed row never consumes an accepted slot.
|
|
3715
|
+
*/
|
|
3716
|
+
function decodeReplyRows(contract, value) {
|
|
3717
|
+
let rawRows;
|
|
3718
|
+
let extras = {};
|
|
3719
|
+
if (contract.parseEnvelope) {
|
|
3720
|
+
const envelope = contract.parseEnvelope(value);
|
|
3721
|
+
rawRows = envelope.rows;
|
|
3722
|
+
extras = envelope.extras;
|
|
3723
|
+
} else {
|
|
3724
|
+
const field = (typeof value === "object" && value !== null && !Array.isArray(value) ? value : void 0)?.[contract.rowsField];
|
|
3725
|
+
if (!Array.isArray(field)) throw new ValidationError(`reply has no "${contract.rowsField}" array`);
|
|
3726
|
+
rawRows = field;
|
|
3727
|
+
}
|
|
3728
|
+
const rows = [];
|
|
3729
|
+
const rejected = [];
|
|
3730
|
+
let overflow = 0;
|
|
3731
|
+
rawRows.forEach((row, index) => {
|
|
3732
|
+
const decoded = contract.decodeRow(row, index);
|
|
3733
|
+
if (!decoded.ok) {
|
|
3734
|
+
rejected.push({
|
|
3735
|
+
index,
|
|
3736
|
+
reason: decoded.reason
|
|
3737
|
+
});
|
|
3738
|
+
return;
|
|
3739
|
+
}
|
|
3740
|
+
if (contract.maxRows !== void 0 && rows.length >= contract.maxRows) {
|
|
3741
|
+
overflow += 1;
|
|
3742
|
+
return;
|
|
3743
|
+
}
|
|
3744
|
+
rows.push(decoded.row);
|
|
3745
|
+
});
|
|
3746
|
+
if (rows.length === 0 && rawRows.length > 0 && contract.whenAllRowsRejected === "fail") throw new ValidationError(`${contract.allRejectedMessage ?? "every reported row was malformed"}: ${rejected.map((entry) => entry.reason).join(" | ")}`);
|
|
3747
|
+
return {
|
|
3748
|
+
rows,
|
|
3749
|
+
extras,
|
|
3750
|
+
rejected,
|
|
3751
|
+
reportedRows: rawRows.length,
|
|
3752
|
+
overflow
|
|
3753
|
+
};
|
|
3754
|
+
}
|
|
3755
|
+
//#endregion
|
|
3597
3756
|
//#region src/analyst/benchmark-public-model.ts
|
|
3598
|
-
/**
|
|
3757
|
+
/** The direct arm as a declarative unit for one public dataset. */
|
|
3758
|
+
function publicDirectAnalystDefinition(dataset, args) {
|
|
3759
|
+
const actor = dataset === "agentrx" ? "agentrx-root-cause-localizer" : "codetracebench-step-localizer";
|
|
3760
|
+
const outputAdapter = dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-block";
|
|
3761
|
+
return {
|
|
3762
|
+
id: "direct",
|
|
3763
|
+
description: "One-shot JSON baseline over the caller-owned model path.",
|
|
3764
|
+
version: "1.0.0",
|
|
3765
|
+
area: dataset === "agentrx" ? "root-cause" : "incorrect",
|
|
3766
|
+
profile: { model: { reasoningEffort: "none" } },
|
|
3767
|
+
question: "",
|
|
3768
|
+
taskDefinition: publicBenchmarkTaskPrompt(dataset),
|
|
3769
|
+
projection: {
|
|
3770
|
+
mode: "chunked",
|
|
3771
|
+
attributeByteCaps: TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS
|
|
3772
|
+
},
|
|
3773
|
+
replyContract: directReplyContract(dataset),
|
|
3774
|
+
contractLimits: dataset === "agentrx" ? { maxFindings: 1 } : {
|
|
3775
|
+
maxBlocks: 16,
|
|
3776
|
+
maxBlockSteps: 12
|
|
3777
|
+
},
|
|
3778
|
+
budget: {
|
|
3779
|
+
timeoutMs: args.timeoutMs,
|
|
3780
|
+
maxCostUsd: args.maxCostUsd,
|
|
3781
|
+
maxOutputTokens: args.maxOutputTokens
|
|
3782
|
+
},
|
|
3783
|
+
repair: { turns: 0 },
|
|
3784
|
+
protocolSha256: publicBenchmarkProtocolSha256(dataset),
|
|
3785
|
+
binding: {
|
|
3786
|
+
kind: "chunked",
|
|
3787
|
+
subjectFromCaseId: (caseId) => trajectoryIdFromCaseId$2(dataset, caseId),
|
|
3788
|
+
baseMetadata: {
|
|
3789
|
+
analysisMode: "direct-baseline",
|
|
3790
|
+
outputAdapter
|
|
3791
|
+
},
|
|
3792
|
+
costActor: actor,
|
|
3793
|
+
costPhase: "analyst.public-benchmark",
|
|
3794
|
+
userMessage: (rendered) => `TRACE DATA:\n${rendered}\n\nReturn the analysis JSON object.`,
|
|
3795
|
+
async expandRows({ subject, rows, store, analystId, producedAt, providerModel, signal }) {
|
|
3796
|
+
const converted = await publicBenchmarkPredictionsToFindings({
|
|
3797
|
+
dataset,
|
|
3798
|
+
trajectoryId: subject,
|
|
3799
|
+
predictions: rows,
|
|
3800
|
+
store,
|
|
3801
|
+
analystId,
|
|
3802
|
+
providerModel: requiredString(providerModel ?? "", "finding providerModel"),
|
|
3803
|
+
producedAt: requiredString(producedAt ?? "", "finding producedAt"),
|
|
3804
|
+
...signal ? { signal } : {}
|
|
3805
|
+
});
|
|
3806
|
+
return {
|
|
3807
|
+
findings: converted.findings,
|
|
3808
|
+
diagnostics: converted.diagnostics
|
|
3809
|
+
};
|
|
3810
|
+
},
|
|
3811
|
+
...dataset === "codetracebench" ? { verifyFindings: async (args) => {
|
|
3812
|
+
await validateCodeTraceFindingEvidence({
|
|
3813
|
+
trajectoryId: args.subject,
|
|
3814
|
+
findings: [...args.findings],
|
|
3815
|
+
store: args.store,
|
|
3816
|
+
...args.signal ? { signal: args.signal } : {}
|
|
3817
|
+
});
|
|
3818
|
+
} } : {}
|
|
3819
|
+
}
|
|
3820
|
+
};
|
|
3821
|
+
}
|
|
3822
|
+
/** Thin shell: validate config, declare the definition, run the chunked strategy. */
|
|
3599
3823
|
function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
3600
3824
|
if (config.instructionsOverride) throw new Error("the direct runner executes only the stock protocol; an instructions override requires the dspy-rlm runner");
|
|
3825
|
+
const maxOutputTokens = positiveSafeInteger(config.maxOutputTokens, "maxOutputTokens");
|
|
3826
|
+
return runChunkedAnalystDefinition(publicDirectAnalystDefinition(dataset, {
|
|
3827
|
+
timeoutMs: positiveSafeInteger(config.timeoutMs, "timeoutMs"),
|
|
3828
|
+
maxOutputTokens,
|
|
3829
|
+
maxCostUsd: config.maxCostUsdPerAnalysis ?? 1
|
|
3830
|
+
}), config);
|
|
3831
|
+
}
|
|
3832
|
+
/**
|
|
3833
|
+
* Compile a chunked-projection definition into a runnable one-shot JSON arm
|
|
3834
|
+
* over the caller-owned model path. Prompt content, the projection ladder,
|
|
3835
|
+
* the reply grammar, and the budget declaration come from the definition;
|
|
3836
|
+
* caching, cost settlement, and the model proxy are transport machinery.
|
|
3837
|
+
*/
|
|
3838
|
+
function runChunkedAnalystDefinition(definition, config) {
|
|
3839
|
+
const { projection, binding, replyContract } = definition;
|
|
3840
|
+
if (projection.mode !== "chunked" || binding.kind !== "chunked") throw new AnalystExpressivenessError(`the chunked one-shot strategy compiles only chunked projections; definition '${definition.id}' declares projection '${projection.mode}' with a '${binding.kind}' binding`);
|
|
3841
|
+
if (config.instructionsOverride) throw new Error("the direct runner executes only the stock protocol; an instructions override requires the dspy-rlm runner");
|
|
3842
|
+
if (definition.repair.turns !== 0) throw new AnalystExpressivenessError(`the one-shot JSON exchange grants no repair turn; definition '${definition.id}' declares ${definition.repair.turns}`);
|
|
3843
|
+
const reasoningEffort = definition.profile.model?.reasoningEffort;
|
|
3844
|
+
if (reasoningEffort !== "none") throw new AnalystExpressivenessError(`the one-shot JSON strategy runs with thinking disabled and can express only reasoning effort 'none'; definition '${definition.id}' declares '${reasoningEffort}'`);
|
|
3845
|
+
if (!replyContract.parseEnvelope) throw new AnalystExpressivenessError(`the one-shot JSON strategy needs a strict reply envelope; definition '${definition.id}' declares no parseEnvelope`);
|
|
3846
|
+
if (definition.taskDefinition === void 0) throw new AnalystExpressivenessError(`the one-shot JSON strategy composes its system prompt from the task definition; definition '${definition.id}' declares none`);
|
|
3601
3847
|
const model = requiredString(config.model, "model");
|
|
3602
3848
|
const callRef = requiredString(config.callRef, "callRef");
|
|
3603
3849
|
if (typeof config.call !== "function") throw new TypeError("call must be a function");
|
|
3604
3850
|
if (typeof config.recordExecution !== "function") throw new TypeError("recordExecution must be a function");
|
|
3605
3851
|
const maxOutputTokens = positiveSafeInteger(config.maxOutputTokens, "maxOutputTokens");
|
|
3606
3852
|
const timeoutMs = positiveSafeInteger(config.timeoutMs, "timeoutMs");
|
|
3853
|
+
const maxCostUsd = config.maxCostUsdPerAnalysis ?? 1;
|
|
3854
|
+
assertDeclaredBudget(definition, {
|
|
3855
|
+
timeoutMs,
|
|
3856
|
+
maxOutputTokens,
|
|
3857
|
+
maxCostUsd
|
|
3858
|
+
});
|
|
3607
3859
|
const maxReasoningTokens = config.maxReasoningTokens ?? maxOutputTokens * 4;
|
|
3608
3860
|
const maxModelRequestBytes = config.maxModelRequestBytes ?? 16 * 1024 * 1024;
|
|
3609
3861
|
const maxModelResponseBytes = config.maxModelResponseBytes ?? 4 * 1024 * 1024;
|
|
3610
3862
|
const modelRequestTimeoutMs = config.modelRequestTimeoutMs ?? timeoutMs;
|
|
3611
3863
|
const pricing = config.pricing ?? pricingForModel$2(model);
|
|
3612
|
-
const maxCostUsd = config.maxCostUsdPerAnalysis ?? 1;
|
|
3613
3864
|
const costLedger = config.costLedger ?? new CostLedger();
|
|
3614
3865
|
const durability = config.durability ? {
|
|
3615
3866
|
runIdentitySha256: requiredString(config.durability.runIdentitySha256, "durability.runIdentitySha256"),
|
|
3616
3867
|
responseCacheDir: requiredString(config.durability.responseCacheDir, "durability.responseCacheDir")
|
|
3617
3868
|
} : void 0;
|
|
3618
|
-
const
|
|
3619
|
-
const outputAdapter = dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-block";
|
|
3869
|
+
const systemPrompt = [definition.taskDefinition, ...replyContract.contractLines].join("\n\n");
|
|
3620
3870
|
return {
|
|
3621
|
-
id:
|
|
3871
|
+
id: definition.id,
|
|
3622
3872
|
async analyze(input, context) {
|
|
3623
|
-
const trajectoryId =
|
|
3873
|
+
const trajectoryId = binding.subjectFromCaseId(context.caseId);
|
|
3624
3874
|
const costTags = {
|
|
3625
|
-
analystId:
|
|
3875
|
+
analystId: binding.costActor,
|
|
3626
3876
|
benchmarkCaseId: context.caseId,
|
|
3627
3877
|
benchmarkRepetition: String(context.repetition)
|
|
3628
3878
|
};
|
|
3629
3879
|
let rawPredictions = [];
|
|
3630
|
-
let
|
|
3880
|
+
let rejectedRows = [];
|
|
3631
3881
|
let modelFindings = [];
|
|
3632
3882
|
let providerModel = model;
|
|
3633
3883
|
let producedAt;
|
|
3634
3884
|
let modelMetadata = {
|
|
3635
|
-
|
|
3636
|
-
|
|
3637
|
-
protocolSha256: publicBenchmarkProtocolSha256(dataset),
|
|
3885
|
+
...binding.baseMetadata,
|
|
3886
|
+
protocolSha256: definition.protocolSha256,
|
|
3638
3887
|
callRef
|
|
3639
3888
|
};
|
|
3640
3889
|
try {
|
|
3641
|
-
if (!input.traceStore) throw new Error(
|
|
3642
|
-
const preparedContext = await prepareSingleTraceContext(input.traceStore, context);
|
|
3643
|
-
if (preparedContext === void 0) throw new Error(
|
|
3890
|
+
if (!input.traceStore) throw new Error(`chunked analyst '${definition.id}' requires a trace store`);
|
|
3891
|
+
const preparedContext = await prepareSingleTraceContext(input.traceStore, context, projection.attributeByteCaps);
|
|
3892
|
+
if (preparedContext === void 0) throw new Error(`trace '${trajectoryId}' has no readable spans`);
|
|
3644
3893
|
const request = {
|
|
3645
3894
|
model,
|
|
3646
3895
|
messages: [{
|
|
3647
3896
|
role: "system",
|
|
3648
|
-
content:
|
|
3897
|
+
content: systemPrompt
|
|
3649
3898
|
}, {
|
|
3650
3899
|
role: "user",
|
|
3651
|
-
content:
|
|
3900
|
+
content: binding.userMessage(preparedContext)
|
|
3652
3901
|
}],
|
|
3653
3902
|
jsonMode: true,
|
|
3654
3903
|
thinking: "disabled",
|
|
@@ -3678,14 +3927,14 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
3678
3927
|
error: cached.error,
|
|
3679
3928
|
metadata: modelMetadata
|
|
3680
3929
|
};
|
|
3681
|
-
const response =
|
|
3682
|
-
rawPredictions = response.
|
|
3683
|
-
|
|
3930
|
+
const response = decodeReplyRows(replyContract, cached.response);
|
|
3931
|
+
rawPredictions = response.rows;
|
|
3932
|
+
rejectedRows = response.rejected.map((entry) => entry.reason);
|
|
3684
3933
|
providerModel = cached.metadata.providerModel;
|
|
3685
3934
|
producedAt = cached.metadata.producedAt;
|
|
3686
3935
|
modelMetadata = {
|
|
3687
3936
|
...modelMetadata,
|
|
3688
|
-
|
|
3937
|
+
...response.extras,
|
|
3689
3938
|
providerModel: cached.metadata.providerModel,
|
|
3690
3939
|
providerDurationMs: cached.metadata.providerDurationMs,
|
|
3691
3940
|
finishReason: cached.metadata.finishReason
|
|
@@ -3714,8 +3963,8 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
3714
3963
|
},
|
|
3715
3964
|
costLedger,
|
|
3716
3965
|
channel: "analyst",
|
|
3717
|
-
phase:
|
|
3718
|
-
actor,
|
|
3966
|
+
phase: binding.costPhase,
|
|
3967
|
+
actor: binding.costActor,
|
|
3719
3968
|
tags: costTags,
|
|
3720
3969
|
callId: providerCallId,
|
|
3721
3970
|
...context.signal ? { signal: context.signal } : {}
|
|
@@ -3734,7 +3983,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
3734
3983
|
...context.signal ? { signal: context.signal } : {},
|
|
3735
3984
|
idempotencyKey: providerCallId
|
|
3736
3985
|
});
|
|
3737
|
-
const response =
|
|
3986
|
+
const response = decodeReplyRows(replyContract, completed.value);
|
|
3738
3987
|
const responseProducedAt = (/* @__PURE__ */ new Date()).toISOString();
|
|
3739
3988
|
const receipt = requiredSettledReceipt(costLedger, providerCallId);
|
|
3740
3989
|
if (cacheIdentity) writePublicBenchmarkResponseCache(durability.responseCacheDir, {
|
|
@@ -3780,26 +4029,25 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
3780
4029
|
}
|
|
3781
4030
|
});
|
|
3782
4031
|
const response = completed.response;
|
|
3783
|
-
rawPredictions = response.
|
|
3784
|
-
|
|
4032
|
+
rawPredictions = response.rows;
|
|
4033
|
+
rejectedRows = response.rejected.map((entry) => entry.reason);
|
|
3785
4034
|
providerModel = completed.result.model;
|
|
3786
4035
|
producedAt = completed.producedAt;
|
|
3787
4036
|
modelMetadata = {
|
|
3788
4037
|
...modelMetadata,
|
|
3789
4038
|
responseSource: "provider",
|
|
3790
|
-
|
|
4039
|
+
...response.extras,
|
|
3791
4040
|
providerModel: completed.result.model,
|
|
3792
4041
|
providerDurationMs: completed.result.durationMs,
|
|
3793
4042
|
finishReason: completed.result.finishReason ?? null,
|
|
3794
4043
|
cost: costReceiptMetadata(completed.receipt)
|
|
3795
4044
|
};
|
|
3796
4045
|
}
|
|
3797
|
-
const converted = await
|
|
3798
|
-
|
|
3799
|
-
|
|
3800
|
-
predictions: rawPredictions,
|
|
4046
|
+
const converted = await binding.expandRows({
|
|
4047
|
+
subject: trajectoryId,
|
|
4048
|
+
rows: rawPredictions,
|
|
3801
4049
|
store: input.traceStore,
|
|
3802
|
-
analystId:
|
|
4050
|
+
analystId: definition.id,
|
|
3803
4051
|
providerModel,
|
|
3804
4052
|
producedAt: requiredString(producedAt ?? "", "finding producedAt"),
|
|
3805
4053
|
...context.signal ? { signal: context.signal } : {}
|
|
@@ -3809,11 +4057,11 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
3809
4057
|
...modelMetadata,
|
|
3810
4058
|
blockDiagnostics: {
|
|
3811
4059
|
...converted.diagnostics,
|
|
3812
|
-
rejectedBlocks
|
|
4060
|
+
rejectedBlocks: rejectedRows
|
|
3813
4061
|
}
|
|
3814
4062
|
};
|
|
3815
|
-
if (
|
|
3816
|
-
trajectoryId,
|
|
4063
|
+
if (binding.verifyFindings) await binding.verifyFindings({
|
|
4064
|
+
subject: trajectoryId,
|
|
3817
4065
|
findings: modelFindings,
|
|
3818
4066
|
store: input.traceStore,
|
|
3819
4067
|
...context.signal ? { signal: context.signal } : {}
|
|
@@ -3846,6 +4094,11 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
3846
4094
|
}
|
|
3847
4095
|
};
|
|
3848
4096
|
}
|
|
4097
|
+
/** A definition that declares one budget while the transport runs another is refused. */
|
|
4098
|
+
function assertDeclaredBudget(definition, effective) {
|
|
4099
|
+
const declared = definition.budget;
|
|
4100
|
+
if (declared.timeoutMs !== effective.timeoutMs || declared.maxOutputTokens !== effective.maxOutputTokens || declared.maxCostUsd !== effective.maxCostUsd) throw new AnalystExpressivenessError(`definition '${definition.id}' declares budget ${JSON.stringify(declared)} but the bound transport runs ${JSON.stringify(effective)}; the declaration must state what executes`);
|
|
4101
|
+
}
|
|
3849
4102
|
function settleCachedResponse(costLedger, cached) {
|
|
3850
4103
|
const settled = costLedger.list().find((receipt) => receipt.callId === cached.callId);
|
|
3851
4104
|
const pending = costLedger.listPending?.().find((record) => record.callId === cached.callId);
|
|
@@ -4003,27 +4256,56 @@ const AgentRxModelResponseSchema = z.object({
|
|
|
4003
4256
|
report: z.string().min(1).max(4e3),
|
|
4004
4257
|
findings: z.array(AgentRxPredictionSchema.extend({ category: AgentRxCategorySchema }).strict()).max(1)
|
|
4005
4258
|
}).strict();
|
|
4006
|
-
|
|
4259
|
+
/**
|
|
4260
|
+
* The one-shot reply grammar per dataset. The envelope is the contract and
|
|
4261
|
+
* stays strict. Individual CodeTraceBench blocks are model output: one
|
|
4262
|
+
* malformed block must not void a case whose remaining blocks are usable and
|
|
4263
|
+
* whose provider call is already paid for, so rows decode individually and
|
|
4264
|
+
* every rejection is reported.
|
|
4265
|
+
*/
|
|
4266
|
+
function directReplyContract(dataset) {
|
|
4007
4267
|
if (dataset === "agentrx") return {
|
|
4008
|
-
|
|
4009
|
-
|
|
4010
|
-
|
|
4011
|
-
|
|
4012
|
-
|
|
4013
|
-
|
|
4014
|
-
|
|
4015
|
-
|
|
4016
|
-
|
|
4017
|
-
|
|
4018
|
-
|
|
4268
|
+
rowsField: "findings",
|
|
4269
|
+
contractLines: [publicBenchmarkFieldContract("agentrx"), PUBLIC_BENCHMARK_ENVELOPE_CONTRACT],
|
|
4270
|
+
repairContractLines: [],
|
|
4271
|
+
parseEnvelope(value) {
|
|
4272
|
+
const parsed = AgentRxModelResponseSchema.parse(value);
|
|
4273
|
+
return {
|
|
4274
|
+
rows: parsed.findings,
|
|
4275
|
+
extras: { report: parsed.report }
|
|
4276
|
+
};
|
|
4277
|
+
},
|
|
4278
|
+
decodeRow(row) {
|
|
4279
|
+
return {
|
|
4280
|
+
ok: true,
|
|
4281
|
+
row
|
|
4282
|
+
};
|
|
4019
4283
|
}
|
|
4020
|
-
|
|
4021
|
-
}
|
|
4022
|
-
if (findings.length === 0 && envelope.findings.length > 0) throw new ValidationError(`every reported failure block was malformed: ${rejectedBlocks.join(" | ")}`);
|
|
4284
|
+
};
|
|
4023
4285
|
return {
|
|
4024
|
-
|
|
4025
|
-
|
|
4026
|
-
|
|
4286
|
+
rowsField: "findings",
|
|
4287
|
+
contractLines: [publicBenchmarkFieldContract("codetracebench"), PUBLIC_BENCHMARK_ENVELOPE_CONTRACT],
|
|
4288
|
+
repairContractLines: [],
|
|
4289
|
+
parseEnvelope(value) {
|
|
4290
|
+
const envelope = CodeTraceModelResponseEnvelopeSchema.parse(value);
|
|
4291
|
+
return {
|
|
4292
|
+
rows: envelope.findings,
|
|
4293
|
+
extras: { report: envelope.report }
|
|
4294
|
+
};
|
|
4295
|
+
},
|
|
4296
|
+
decodeRow(row, index) {
|
|
4297
|
+
const parsed = CodeTraceBlockPredictionSchema.safeParse(row);
|
|
4298
|
+
if (parsed.success) return {
|
|
4299
|
+
ok: true,
|
|
4300
|
+
row: parsed.data
|
|
4301
|
+
};
|
|
4302
|
+
return {
|
|
4303
|
+
ok: false,
|
|
4304
|
+
reason: `block ${index}: ${parsed.error.issues.map((issue) => `${issue.path.join(".") || "<root>"} ${issue.message}`).join("; ")}`
|
|
4305
|
+
};
|
|
4306
|
+
},
|
|
4307
|
+
whenAllRowsRejected: "fail",
|
|
4308
|
+
allRejectedMessage: "every reported failure block was malformed"
|
|
4027
4309
|
};
|
|
4028
4310
|
}
|
|
4029
4311
|
async function publicBenchmarkPredictionsToFindings(options) {
|
|
@@ -4090,12 +4372,12 @@ async function publicBenchmarkPredictionsToFindings(options) {
|
|
|
4090
4372
|
...options.signal ? { signal: options.signal } : {}
|
|
4091
4373
|
});
|
|
4092
4374
|
}
|
|
4093
|
-
async function prepareSingleTraceContext(store, context) {
|
|
4375
|
+
async function prepareSingleTraceContext(store, context, attributeByteCaps) {
|
|
4094
4376
|
const storeContext = context.signal ? { signal: context.signal } : void 0;
|
|
4095
4377
|
const overview = await store.getOverview(void 0, storeContext);
|
|
4096
4378
|
if (overview.total_traces !== 1 || overview.sample_trace_ids.length !== 1) throw new Error(`public model benchmark requires exactly one trace, received ${overview.total_traces}`);
|
|
4097
4379
|
const traceId = overview.sample_trace_ids[0];
|
|
4098
|
-
for (const perAttributeByteCap of
|
|
4380
|
+
for (const perAttributeByteCap of attributeByteCaps) {
|
|
4099
4381
|
const viewed = await store.viewTrace({
|
|
4100
4382
|
trace_id: traceId,
|
|
4101
4383
|
per_attribute_byte_cap: perAttributeByteCap
|
|
@@ -4245,18 +4527,156 @@ function contiguousSegments(sortedSteps) {
|
|
|
4245
4527
|
}
|
|
4246
4528
|
//#endregion
|
|
4247
4529
|
//#region src/analyst/benchmark-public-rlm.ts
|
|
4248
|
-
/**
|
|
4530
|
+
/** The dspy-rlm arm as a declarative unit for one public dataset. */
|
|
4531
|
+
function publicRlmAnalystDefinition(dataset, args) {
|
|
4532
|
+
return {
|
|
4533
|
+
id: "dspy-rlm",
|
|
4534
|
+
description: dataset === "agentrx" ? "Localizes the first unrecoverable root-cause step." : "Localizes every incorrect state-changing assistant step.",
|
|
4535
|
+
version: "1.0.0",
|
|
4536
|
+
area: dataset === "agentrx" ? "root-cause" : "incorrect",
|
|
4537
|
+
profile: {},
|
|
4538
|
+
question: dataset === "agentrx" ? "What is the first unrecoverable root cause in this failed trajectory?" : "Which assistant steps are incorrect under the CodeTraceBench definition?",
|
|
4539
|
+
taskDefinition: args.instructions,
|
|
4540
|
+
projection: {
|
|
4541
|
+
mode: "repl-variable",
|
|
4542
|
+
toolGroup: "singleTrace"
|
|
4543
|
+
},
|
|
4544
|
+
replyContract: {
|
|
4545
|
+
rowsField: "findings",
|
|
4546
|
+
contractLines: [RAW_FINDING_SCHEMA_PROMPT],
|
|
4547
|
+
repairContractLines: [],
|
|
4548
|
+
decodeRow(row) {
|
|
4549
|
+
const parsed = RawAnalystFindingSchema.safeParse(row);
|
|
4550
|
+
if (parsed.success) return {
|
|
4551
|
+
ok: true,
|
|
4552
|
+
row: parsed.data
|
|
4553
|
+
};
|
|
4554
|
+
return {
|
|
4555
|
+
ok: false,
|
|
4556
|
+
reason: parsed.error.issues.map((issue) => `${issue.path.join(".")}: ${issue.message}`).join("; ")
|
|
4557
|
+
};
|
|
4558
|
+
}
|
|
4559
|
+
},
|
|
4560
|
+
contractLimits: {
|
|
4561
|
+
maxIterations: args.engineLimits.maxIterations,
|
|
4562
|
+
maxLlmCalls: args.engineLimits.maxLlmCalls,
|
|
4563
|
+
maxToolCalls: args.engineLimits.maxToolCalls,
|
|
4564
|
+
maxOutputChars: args.engineLimits.maxOutputChars
|
|
4565
|
+
},
|
|
4566
|
+
budget: {
|
|
4567
|
+
timeoutMs: args.timeoutMs,
|
|
4568
|
+
maxCostUsd: args.maxCostUsd,
|
|
4569
|
+
maxOutputTokens: args.maxOutputTokens,
|
|
4570
|
+
engineLimits: args.engineLimits
|
|
4571
|
+
},
|
|
4572
|
+
repair: { turns: 1 },
|
|
4573
|
+
protocolSha256: args.protocolSha256,
|
|
4574
|
+
binding: {
|
|
4575
|
+
kind: "repl-variable",
|
|
4576
|
+
traceAnalystId: dataset === "agentrx" ? "agentrx-dspy-rlm" : "codetracebench-dspy-rlm",
|
|
4577
|
+
subjectFromCaseId: (caseId) => trajectoryIdFromCaseId$1(dataset, caseId),
|
|
4578
|
+
baseMetadata: {
|
|
4579
|
+
analysisMode: "recursive",
|
|
4580
|
+
engine: "dspy-rlm"
|
|
4581
|
+
},
|
|
4582
|
+
findingBaseMetadata: {
|
|
4583
|
+
analysis_mode: "recursive",
|
|
4584
|
+
engine: "dspy-rlm"
|
|
4585
|
+
},
|
|
4586
|
+
costPhase: "analyst.public-benchmark.dspy-rlm",
|
|
4587
|
+
...dataset === "codetracebench" ? { metadataFromSubject: codeTraceBlockMetadataFromSubject } : {},
|
|
4588
|
+
async adapt({ subject, findings, analystId, store, signal }) {
|
|
4589
|
+
return adaptPublicBenchmarkFindings({
|
|
4590
|
+
dataset,
|
|
4591
|
+
trajectoryId: subject,
|
|
4592
|
+
findings: [...findings],
|
|
4593
|
+
analystId,
|
|
4594
|
+
store,
|
|
4595
|
+
...signal ? { signal } : {}
|
|
4596
|
+
});
|
|
4597
|
+
},
|
|
4598
|
+
...dataset === "codetracebench" ? { consensus: codeTraceConsensusPort() } : {},
|
|
4599
|
+
abstentionFallback: (fallbackConfig) => createPublicBenchmarkDirectRunner(dataset, fallbackConfig)
|
|
4600
|
+
}
|
|
4601
|
+
};
|
|
4602
|
+
}
|
|
4603
|
+
/** Step-level majority consensus on the CodeTraceBench block grammar. */
|
|
4604
|
+
function codeTraceConsensusPort() {
|
|
4605
|
+
return {
|
|
4606
|
+
vote(samples) {
|
|
4607
|
+
const consensus = consensusCodeTraceBlocks(samples.map((sample) => [...sample]));
|
|
4608
|
+
return {
|
|
4609
|
+
blocks: consensus.blocks,
|
|
4610
|
+
decision: consensus.decision
|
|
4611
|
+
};
|
|
4612
|
+
},
|
|
4613
|
+
async expand({ subject, blocks, store, analystId, producedAt, signal }) {
|
|
4614
|
+
const expanded = await expandCodeTraceFailureBlocks({
|
|
4615
|
+
trajectoryId: subject,
|
|
4616
|
+
blocks,
|
|
4617
|
+
store,
|
|
4618
|
+
analystId,
|
|
4619
|
+
producedAt,
|
|
4620
|
+
...signal ? { signal } : {}
|
|
4621
|
+
});
|
|
4622
|
+
return {
|
|
4623
|
+
findings: expanded.findings,
|
|
4624
|
+
diagnostics: expanded.diagnostics
|
|
4625
|
+
};
|
|
4626
|
+
},
|
|
4627
|
+
sampleRecord(assignments) {
|
|
4628
|
+
return {
|
|
4629
|
+
blocks: sampleBlockRecords(assignments),
|
|
4630
|
+
steps: assignments.map((assignment) => assignment.step)
|
|
4631
|
+
};
|
|
4632
|
+
}
|
|
4633
|
+
};
|
|
4634
|
+
}
|
|
4635
|
+
/** Thin shell: validate config, declare the definition, run the repl-variable strategy. */
|
|
4249
4636
|
function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
4250
|
-
const costLedger = config.costLedger ?? new CostLedger();
|
|
4251
4637
|
const samples = config.dspyRlm?.samples ?? 1;
|
|
4252
4638
|
if (!Number.isSafeInteger(samples) || samples < 1) throw new RangeError("dspyRlm.samples must be a positive safe integer");
|
|
4253
4639
|
if (samples > 1 && dataset !== "codetracebench") throw new Error("dspyRlm.samples > 1 requires the codetracebench dataset; step-level consensus is defined on its block grammar");
|
|
4254
|
-
|
|
4640
|
+
return runReplVariableAnalystDefinition(publicRlmAnalystDefinition(dataset, {
|
|
4641
|
+
instructions: config.instructionsOverride?.text ?? publicBenchmarkRlmInstructions(dataset),
|
|
4642
|
+
protocolSha256: effectiveAnalystProtocolSha256(dataset, config.instructionsOverride),
|
|
4643
|
+
timeoutMs: config.timeoutMs,
|
|
4644
|
+
maxOutputTokens: config.maxOutputTokens,
|
|
4645
|
+
maxCostUsd: config.maxCostUsdPerAnalysis ?? 1,
|
|
4646
|
+
engineLimits: rlmEngineLimits(config)
|
|
4647
|
+
}), config);
|
|
4648
|
+
}
|
|
4649
|
+
/** Engine iteration limits with this arm's defaults applied. */
|
|
4650
|
+
function rlmEngineLimits(config) {
|
|
4651
|
+
return {
|
|
4255
4652
|
maxIterations: config.dspyRlm?.maxIterations ?? 14,
|
|
4256
4653
|
maxLlmCalls: config.dspyRlm?.maxLlmCalls ?? 8,
|
|
4257
4654
|
maxToolCalls: config.dspyRlm?.maxToolCalls ?? 80,
|
|
4258
4655
|
maxOutputChars: config.dspyRlm?.maxOutputChars ?? 8e3
|
|
4259
4656
|
};
|
|
4657
|
+
}
|
|
4658
|
+
/**
|
|
4659
|
+
* Compile a repl-variable definition into a runnable recursive-engine arm over
|
|
4660
|
+
* the caller-owned model path. The question, instructions, tool group, and
|
|
4661
|
+
* iteration limits come from the definition; the engine, model proxy, sampling
|
|
4662
|
+
* loop, and abstention floor are transport machinery.
|
|
4663
|
+
*/
|
|
4664
|
+
function runReplVariableAnalystDefinition(definition, config) {
|
|
4665
|
+
const { projection, binding } = definition;
|
|
4666
|
+
if (projection.mode !== "repl-variable" || binding.kind !== "repl-variable") throw new AnalystExpressivenessError(`the repl-variable strategy compiles only repl-variable projections; definition '${definition.id}' declares projection '${projection.mode}' with a '${binding.kind}' binding`);
|
|
4667
|
+
const instructions = definition.taskDefinition;
|
|
4668
|
+
if (instructions === void 0) throw new AnalystExpressivenessError(`the repl-variable strategy runs the definition's task text as engine instructions; definition '${definition.id}' declares none`);
|
|
4669
|
+
if (config.instructionsOverride !== void 0 && config.instructionsOverride.text !== instructions) throw new AnalystExpressivenessError(`definition '${definition.id}' declares instructions that differ from the transport's instructionsOverride; one text must execute`);
|
|
4670
|
+
const area = definition.area;
|
|
4671
|
+
if (area === void 0) throw new AnalystExpressivenessError(`the repl-variable strategy stamps the definition's area on every finding; definition '${definition.id}' declares none`);
|
|
4672
|
+
const limits = definition.budget.engineLimits;
|
|
4673
|
+
if (limits === void 0) throw new AnalystExpressivenessError(`the repl-variable strategy needs declared engine limits; definition '${definition.id}' declares none`);
|
|
4674
|
+
const effectiveLimits = rlmEngineLimits(config);
|
|
4675
|
+
if (limits.maxIterations !== effectiveLimits.maxIterations || limits.maxLlmCalls !== effectiveLimits.maxLlmCalls || limits.maxToolCalls !== effectiveLimits.maxToolCalls || limits.maxOutputChars !== effectiveLimits.maxOutputChars) throw new AnalystExpressivenessError(`definition '${definition.id}' declares engine limits ${JSON.stringify(limits)} but the bound transport runs ${JSON.stringify(effectiveLimits)}; the declaration must state what executes`);
|
|
4676
|
+
const costLedger = config.costLedger ?? new CostLedger();
|
|
4677
|
+
const samples = config.dspyRlm?.samples ?? 1;
|
|
4678
|
+
if (!Number.isSafeInteger(samples) || samples < 1) throw new RangeError("dspyRlm.samples must be a positive safe integer");
|
|
4679
|
+
if (samples > 1 && binding.consensus === void 0) throw new AnalystExpressivenessError(`samples > 1 needs a consensus port; definition '${definition.id}' declares none`);
|
|
4260
4680
|
const pricing = config.pricing ?? pricingForModel$1(config.model);
|
|
4261
4681
|
const engine = createDspyRlmTraceEngine({
|
|
4262
4682
|
call: config.call,
|
|
@@ -4279,18 +4699,26 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4279
4699
|
...config.dspyRlm?.traceToolTimeoutMs === void 0 ? {} : { traceToolTimeoutMs: config.dspyRlm.traceToolTimeoutMs },
|
|
4280
4700
|
...config.dspyRlm?.runner ? { runner: config.dspyRlm.runner } : {}
|
|
4281
4701
|
});
|
|
4282
|
-
const
|
|
4283
|
-
const
|
|
4284
|
-
|
|
4702
|
+
const protocolSha256 = definition.protocolSha256;
|
|
4703
|
+
const traceDefinition = {
|
|
4704
|
+
id: binding.traceAnalystId,
|
|
4705
|
+
description: definition.description,
|
|
4706
|
+
area,
|
|
4707
|
+
version: definition.version,
|
|
4708
|
+
question: definition.question,
|
|
4709
|
+
instructions,
|
|
4710
|
+
toolGroup: projection.toolGroup,
|
|
4711
|
+
limits
|
|
4712
|
+
};
|
|
4285
4713
|
const { instructionsOverride: _rlmOnlyOverride, ...directConfig } = config;
|
|
4286
|
-
const abstentionFallbackRunner =
|
|
4714
|
+
const abstentionFallbackRunner = binding.abstentionFallback({
|
|
4287
4715
|
...directConfig,
|
|
4288
4716
|
costLedger
|
|
4289
4717
|
});
|
|
4290
4718
|
return {
|
|
4291
|
-
id:
|
|
4719
|
+
id: definition.id,
|
|
4292
4720
|
async analyze(input, context) {
|
|
4293
|
-
const trajectoryId =
|
|
4721
|
+
const trajectoryId = binding.subjectFromCaseId(context.caseId);
|
|
4294
4722
|
const tags = {
|
|
4295
4723
|
benchmarkCaseId: context.caseId,
|
|
4296
4724
|
benchmarkRepetition: String(context.repetition)
|
|
@@ -4298,7 +4726,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4298
4726
|
let usage;
|
|
4299
4727
|
let rawFindings = [];
|
|
4300
4728
|
try {
|
|
4301
|
-
if (!input.traceStore) throw new Error(
|
|
4729
|
+
if (!input.traceStore) throw new Error(`repl-variable analyst '${definition.id}' requires a trace store`);
|
|
4302
4730
|
if (samples > 1) {
|
|
4303
4731
|
const store = input.traceStore;
|
|
4304
4732
|
const caseUsageFilter = {
|
|
@@ -4312,14 +4740,14 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4312
4740
|
for (let sample = 0; sample < samples; sample += 1) {
|
|
4313
4741
|
let sampleUsage;
|
|
4314
4742
|
const completed = await runTraceAnalyst({
|
|
4315
|
-
definition,
|
|
4743
|
+
definition: traceDefinition,
|
|
4316
4744
|
engine,
|
|
4317
4745
|
store,
|
|
4318
4746
|
context: {
|
|
4319
4747
|
runId: context.caseId,
|
|
4320
4748
|
correlationId: `${context.caseId}:${context.repetition}:sample-${sample}`,
|
|
4321
4749
|
costLedger,
|
|
4322
|
-
costPhase:
|
|
4750
|
+
costPhase: binding.costPhase,
|
|
4323
4751
|
tags,
|
|
4324
4752
|
recordUsage: (receipt) => {
|
|
4325
4753
|
sampleUsage = receipt;
|
|
@@ -4330,8 +4758,8 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4330
4758
|
});
|
|
4331
4759
|
const producedAt = (/* @__PURE__ */ new Date()).toISOString();
|
|
4332
4760
|
const sampleFindings = completed.findings.map((finding) => makeFinding({
|
|
4333
|
-
analyst_id:
|
|
4334
|
-
area
|
|
4761
|
+
analyst_id: definition.id,
|
|
4762
|
+
area,
|
|
4335
4763
|
subject: finding.subject,
|
|
4336
4764
|
claim: finding.claim,
|
|
4337
4765
|
rationale: finding.rationale,
|
|
@@ -4340,20 +4768,18 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4340
4768
|
evidence_refs: evidenceRefsFromRawFinding(finding),
|
|
4341
4769
|
recommended_action: finding.recommended_action,
|
|
4342
4770
|
metadata: {
|
|
4343
|
-
|
|
4344
|
-
engine: "dspy-rlm",
|
|
4771
|
+
...binding.findingBaseMetadata,
|
|
4345
4772
|
model: config.model,
|
|
4346
4773
|
sample,
|
|
4347
|
-
...
|
|
4774
|
+
...binding.metadataFromSubject?.(finding.subject) ?? {}
|
|
4348
4775
|
},
|
|
4349
4776
|
produced_at: producedAt
|
|
4350
4777
|
}));
|
|
4351
4778
|
rawFindings = [...rawFindings, ...sampleFindings];
|
|
4352
|
-
const adapted = await
|
|
4353
|
-
|
|
4354
|
-
trajectoryId,
|
|
4779
|
+
const adapted = await binding.adapt({
|
|
4780
|
+
subject: trajectoryId,
|
|
4355
4781
|
findings: sampleFindings,
|
|
4356
|
-
analystId:
|
|
4782
|
+
analystId: definition.id,
|
|
4357
4783
|
store,
|
|
4358
4784
|
...context.signal ? { signal: context.signal } : {}
|
|
4359
4785
|
});
|
|
@@ -4368,18 +4794,17 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4368
4794
|
modelCalls: completed.modelCalls,
|
|
4369
4795
|
toolCalls: completed.toolCalls,
|
|
4370
4796
|
runtime: completed.runtime,
|
|
4371
|
-
|
|
4372
|
-
steps: assignments.map((assignment) => assignment.step),
|
|
4797
|
+
...binding.consensus.sampleRecord(assignments),
|
|
4373
4798
|
...adapted.diagnostics ? { blockDiagnostics: adapted.diagnostics } : {},
|
|
4374
4799
|
...sampleUsage ? { usage: sampleUsage } : {}
|
|
4375
4800
|
});
|
|
4376
4801
|
}
|
|
4377
|
-
const consensus =
|
|
4378
|
-
const expanded = await
|
|
4379
|
-
trajectoryId,
|
|
4802
|
+
const consensus = binding.consensus.vote(sampleAssignments);
|
|
4803
|
+
const expanded = await binding.consensus.expand({
|
|
4804
|
+
subject: trajectoryId,
|
|
4380
4805
|
blocks: consensus.blocks,
|
|
4381
4806
|
store,
|
|
4382
|
-
analystId:
|
|
4807
|
+
analystId: definition.id,
|
|
4383
4808
|
producedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
4384
4809
|
...context.signal ? { signal: context.signal } : {}
|
|
4385
4810
|
});
|
|
@@ -4390,8 +4815,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4390
4815
|
findings: fallback && !fallback.error ? fallback.findings : expanded.findings,
|
|
4391
4816
|
usage,
|
|
4392
4817
|
metadata: {
|
|
4393
|
-
|
|
4394
|
-
engine: "dspy-rlm",
|
|
4818
|
+
...binding.baseMetadata,
|
|
4395
4819
|
protocolSha256,
|
|
4396
4820
|
samples,
|
|
4397
4821
|
sampleRuns,
|
|
@@ -4408,14 +4832,14 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4408
4832
|
};
|
|
4409
4833
|
}
|
|
4410
4834
|
const completed = await runTraceAnalyst({
|
|
4411
|
-
definition,
|
|
4835
|
+
definition: traceDefinition,
|
|
4412
4836
|
engine,
|
|
4413
4837
|
store: input.traceStore,
|
|
4414
4838
|
context: {
|
|
4415
4839
|
runId: context.caseId,
|
|
4416
4840
|
correlationId: `${context.caseId}:${context.repetition}`,
|
|
4417
4841
|
costLedger,
|
|
4418
|
-
costPhase:
|
|
4842
|
+
costPhase: binding.costPhase,
|
|
4419
4843
|
tags,
|
|
4420
4844
|
recordUsage: (receipt) => {
|
|
4421
4845
|
usage = receipt;
|
|
@@ -4425,8 +4849,8 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4425
4849
|
});
|
|
4426
4850
|
const producedAt = (/* @__PURE__ */ new Date()).toISOString();
|
|
4427
4851
|
rawFindings = completed.findings.map((finding) => makeFinding({
|
|
4428
|
-
analyst_id:
|
|
4429
|
-
area
|
|
4852
|
+
analyst_id: definition.id,
|
|
4853
|
+
area,
|
|
4430
4854
|
subject: finding.subject,
|
|
4431
4855
|
claim: finding.claim,
|
|
4432
4856
|
rationale: finding.rationale,
|
|
@@ -4435,18 +4859,16 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4435
4859
|
evidence_refs: evidenceRefsFromRawFinding(finding),
|
|
4436
4860
|
recommended_action: finding.recommended_action,
|
|
4437
4861
|
metadata: {
|
|
4438
|
-
|
|
4439
|
-
engine: "dspy-rlm",
|
|
4862
|
+
...binding.findingBaseMetadata,
|
|
4440
4863
|
model: config.model,
|
|
4441
|
-
...
|
|
4864
|
+
...binding.metadataFromSubject?.(finding.subject) ?? {}
|
|
4442
4865
|
},
|
|
4443
4866
|
produced_at: producedAt
|
|
4444
4867
|
}));
|
|
4445
|
-
const adapted = await
|
|
4446
|
-
|
|
4447
|
-
trajectoryId,
|
|
4868
|
+
const adapted = await binding.adapt({
|
|
4869
|
+
subject: trajectoryId,
|
|
4448
4870
|
findings: rawFindings,
|
|
4449
|
-
analystId:
|
|
4871
|
+
analystId: definition.id,
|
|
4450
4872
|
store: input.traceStore,
|
|
4451
4873
|
...context.signal ? { signal: context.signal } : {}
|
|
4452
4874
|
});
|
|
@@ -4465,8 +4887,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4465
4887
|
findings: fallback && !fallback.error ? fallback.findings : adapted.findings,
|
|
4466
4888
|
usage,
|
|
4467
4889
|
metadata: {
|
|
4468
|
-
|
|
4469
|
-
engine: "dspy-rlm",
|
|
4890
|
+
...binding.baseMetadata,
|
|
4470
4891
|
protocolSha256,
|
|
4471
4892
|
...adapted.diagnostics ? { blockDiagnostics: adapted.diagnostics } : {},
|
|
4472
4893
|
answer: completed.answer,
|
|
@@ -4489,8 +4910,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4489
4910
|
usage,
|
|
4490
4911
|
error: publicBenchmarkError(error, []),
|
|
4491
4912
|
metadata: {
|
|
4492
|
-
|
|
4493
|
-
engine: "dspy-rlm",
|
|
4913
|
+
...binding.baseMetadata,
|
|
4494
4914
|
...samples > 1 ? { samples } : {},
|
|
4495
4915
|
rawFindings
|
|
4496
4916
|
}
|
|
@@ -4518,18 +4938,6 @@ function sampleBlockRecords(assignments) {
|
|
|
4518
4938
|
acceptedSteps
|
|
4519
4939
|
}));
|
|
4520
4940
|
}
|
|
4521
|
-
function publicBenchmarkDefinition(dataset, limits, instructions) {
|
|
4522
|
-
return {
|
|
4523
|
-
id: dataset === "agentrx" ? "agentrx-dspy-rlm" : "codetracebench-dspy-rlm",
|
|
4524
|
-
description: dataset === "agentrx" ? "Localizes the first unrecoverable root-cause step." : "Localizes every incorrect state-changing assistant step.",
|
|
4525
|
-
area: dataset === "agentrx" ? "root-cause" : "incorrect",
|
|
4526
|
-
version: "1.0.0",
|
|
4527
|
-
question: dataset === "agentrx" ? "What is the first unrecoverable root cause in this failed trajectory?" : "Which assistant steps are incorrect under the CodeTraceBench definition?",
|
|
4528
|
-
instructions,
|
|
4529
|
-
toolGroup: "singleTrace",
|
|
4530
|
-
limits
|
|
4531
|
-
};
|
|
4532
|
-
}
|
|
4533
4941
|
function pricingForModel$1(model) {
|
|
4534
4942
|
const pricing = resolveModelPricing(model);
|
|
4535
4943
|
if (!pricing) throw new Error(`no pricing is configured for '${model}'; provide PublicAnalystBenchmarkModelConfig.pricing`);
|
|
@@ -4893,455 +5301,31 @@ function nodeHttpPrimeBridgeTransport() {
|
|
|
4893
5301
|
};
|
|
4894
5302
|
}
|
|
4895
5303
|
//#endregion
|
|
4896
|
-
//#region src/analyst/prime-protocol.ts
|
|
4897
|
-
function buildPrimePrompt(spec) {
|
|
4898
|
-
return [
|
|
4899
|
-
`QUESTION: ${spec.question}`,
|
|
4900
|
-
"",
|
|
4901
|
-
...spec.taskDefinition === void 0 ? [] : [
|
|
4902
|
-
"TASK DEFINITION:",
|
|
4903
|
-
spec.taskDefinition,
|
|
4904
|
-
""
|
|
4905
|
-
],
|
|
4906
|
-
...spec.contractLines,
|
|
4907
|
-
"",
|
|
4908
|
-
spec.trajectoryHeader,
|
|
4909
|
-
spec.renderedTrajectory,
|
|
4910
|
-
...spec.trailer === void 0 ? [] : ["", spec.trailer]
|
|
4911
|
-
].join("\n");
|
|
4912
|
-
}
|
|
4913
|
-
/** Carries the malformed reply and the contract — never the trajectory. */
|
|
4914
|
-
function buildPrimeRepairPrompt(spec) {
|
|
4915
|
-
return [
|
|
4916
|
-
"Your previous reply to a trace-analysis task was structurally malformed and could not be parsed",
|
|
4917
|
-
`(${spec.defect}). Below is your previous reply verbatim. Re-emit ONLY the corrected JSON — one`,
|
|
4918
|
-
"fenced ```json block, no other text, no tools. The JSON object has exactly two fields:",
|
|
4919
|
-
...spec.repairContractLines,
|
|
4920
|
-
"",
|
|
4921
|
-
"PREVIOUS REPLY:",
|
|
4922
|
-
spec.previousReply
|
|
4923
|
-
].join("\n");
|
|
4924
|
-
}
|
|
4925
|
-
/**
|
|
4926
|
-
* Recover the reply's JSON object.
|
|
4927
|
-
*
|
|
4928
|
-
* Distinct from `extractJsonPayload` in ../llm-client, which serves a response
|
|
4929
|
-
* that DECLARES a JSON root and therefore must not scan onward. A prime reply
|
|
4930
|
-
* is prose plus a fenced block, and when the model emits several fences the
|
|
4931
|
-
* last one is its answer — so fences are scanned in reverse, and only then is a
|
|
4932
|
-
* brace-to-brace slice tried.
|
|
4933
|
-
*/
|
|
4934
|
-
function extractPrimeJsonObject(text) {
|
|
4935
|
-
const direct = parsePrimeJsonObject(text);
|
|
4936
|
-
if (direct) return direct;
|
|
4937
|
-
const fenced = [...text.matchAll(/```(?:json)?\s*\n?([\s\S]*?)```/g)];
|
|
4938
|
-
for (let index = fenced.length - 1; index >= 0; index -= 1) {
|
|
4939
|
-
const candidate = parsePrimeJsonObject(fenced[index][1]);
|
|
4940
|
-
if (candidate) return candidate;
|
|
4941
|
-
}
|
|
4942
|
-
const start = text.indexOf("{");
|
|
4943
|
-
const end = text.lastIndexOf("}");
|
|
4944
|
-
if (start >= 0 && end > start) {
|
|
4945
|
-
const candidate = parsePrimeJsonObject(text.slice(start, end + 1));
|
|
4946
|
-
if (candidate) return candidate;
|
|
4947
|
-
}
|
|
4948
|
-
return null;
|
|
4949
|
-
}
|
|
4950
|
-
/** Why the reply cannot be read as a prime answer, or null when it can. */
|
|
4951
|
-
function primeReplyDefect(parsed, rowsField) {
|
|
4952
|
-
if (parsed === null) return "no parseable JSON object";
|
|
4953
|
-
if (!Array.isArray(parsed[rowsField])) return `JSON has no "${rowsField}" array`;
|
|
4954
|
-
return null;
|
|
4955
|
-
}
|
|
4956
|
-
function parsePrimeJsonObject(text) {
|
|
4957
|
-
try {
|
|
4958
|
-
const value = JSON.parse(text.trim());
|
|
4959
|
-
return typeof value === "object" && value !== null && !Array.isArray(value) ? value : null;
|
|
4960
|
-
} catch {
|
|
4961
|
-
return null;
|
|
4962
|
-
}
|
|
4963
|
-
}
|
|
4964
|
-
function emptyPrimeRawUsage() {
|
|
4965
|
-
return {
|
|
4966
|
-
calls: null,
|
|
4967
|
-
inputTokens: null,
|
|
4968
|
-
outputTokens: null,
|
|
4969
|
-
bridgeEstimated: false
|
|
4970
|
-
};
|
|
4971
|
-
}
|
|
4972
|
-
/** Read the bridge's OpenAI-shaped `usage` object. */
|
|
4973
|
-
function normalizePrimeUsage(raw) {
|
|
4974
|
-
if (typeof raw !== "object" || raw === null) return emptyPrimeRawUsage();
|
|
4975
|
-
const record = raw;
|
|
4976
|
-
return {
|
|
4977
|
-
calls: tokenCountOrNull(record.model_requests),
|
|
4978
|
-
inputTokens: tokenCountOrNull(record.prompt_tokens),
|
|
4979
|
-
outputTokens: tokenCountOrNull(record.completion_tokens),
|
|
4980
|
-
bridgeEstimated: record.estimated === true
|
|
4981
|
-
};
|
|
4982
|
-
}
|
|
4983
|
-
/**
|
|
4984
|
-
* Sum two turns. Each side poisons independently: two turns that both report
|
|
4985
|
-
* input and neither report output yield a real input total beside a null
|
|
4986
|
-
* output, because discarding a measured count is as wrong as inventing one.
|
|
4987
|
-
*/
|
|
4988
|
-
function mergePrimeRawUsage(a, b) {
|
|
4989
|
-
return {
|
|
4990
|
-
calls: sumOrNull(a.calls, b.calls),
|
|
4991
|
-
inputTokens: sumOrNull(a.inputTokens, b.inputTokens),
|
|
4992
|
-
outputTokens: sumOrNull(a.outputTokens, b.outputTokens),
|
|
4993
|
-
bridgeEstimated: a.bridgeEstimated || b.bridgeEstimated
|
|
4994
|
-
};
|
|
4995
|
-
}
|
|
4996
|
-
function sumOrNull(a, b) {
|
|
4997
|
-
return a !== null && b !== null ? a + b : null;
|
|
4998
|
-
}
|
|
4999
|
-
function tokenCountOrNull(value) {
|
|
5000
|
-
return typeof value === "number" && Number.isSafeInteger(value) && value >= 0 ? value : null;
|
|
5001
|
-
}
|
|
5002
|
-
/**
|
|
5003
|
-
* Bind raw prime usage to agent-eval's typed receipt.
|
|
5004
|
-
*
|
|
5005
|
-
* `RunTokenUsage.input` and `.output` are non-nullable, so a one-sided count
|
|
5006
|
-
* cannot round-trip through `tokens` without writing a zero nobody measured.
|
|
5007
|
-
* The complete-accounting field therefore stays null, the reported side is
|
|
5008
|
-
* carried verbatim in `partialTokens`, and its price becomes the receipt's
|
|
5009
|
-
* `knownCostUsd` lower bound.
|
|
5010
|
-
*
|
|
5011
|
-
* Only agent-eval calls this; consumers with no pricing table read
|
|
5012
|
-
* `PrimeRawUsage` directly.
|
|
5013
|
-
*/
|
|
5014
|
-
function analystUsageReceiptFromPrimeUsage(usage, pricing) {
|
|
5015
|
-
const { calls, inputTokens, outputTokens, bridgeEstimated } = usage;
|
|
5016
|
-
const estimatedTokens = bridgeEstimated ? { tokensEstimated: true } : {};
|
|
5017
|
-
if (inputTokens !== null && outputTokens !== null) return {
|
|
5018
|
-
calls,
|
|
5019
|
-
tokens: {
|
|
5020
|
-
input: inputTokens,
|
|
5021
|
-
output: outputTokens
|
|
5022
|
-
},
|
|
5023
|
-
cost: {
|
|
5024
|
-
kind: "estimated",
|
|
5025
|
-
usd: priceTokens(inputTokens, outputTokens, pricing)
|
|
5026
|
-
},
|
|
5027
|
-
...estimatedTokens
|
|
5028
|
-
};
|
|
5029
|
-
if (inputTokens === null && outputTokens === null) return {
|
|
5030
|
-
calls,
|
|
5031
|
-
tokens: null,
|
|
5032
|
-
cost: {
|
|
5033
|
-
kind: "uncaptured",
|
|
5034
|
-
usd: null
|
|
5035
|
-
},
|
|
5036
|
-
...estimatedTokens
|
|
5037
|
-
};
|
|
5038
|
-
return {
|
|
5039
|
-
calls,
|
|
5040
|
-
tokens: null,
|
|
5041
|
-
partialTokens: {
|
|
5042
|
-
input: inputTokens,
|
|
5043
|
-
output: outputTokens
|
|
5044
|
-
},
|
|
5045
|
-
cost: {
|
|
5046
|
-
kind: "uncaptured",
|
|
5047
|
-
usd: null
|
|
5048
|
-
},
|
|
5049
|
-
knownCostUsd: priceTokens(inputTokens ?? 0, outputTokens ?? 0, pricing),
|
|
5050
|
-
...estimatedTokens
|
|
5051
|
-
};
|
|
5052
|
-
}
|
|
5053
|
-
function priceTokens(input, output, pricing) {
|
|
5054
|
-
return (input * pricing.inputUsdPerMillion + output * pricing.outputUsdPerMillion) / 1e6;
|
|
5055
|
-
}
|
|
5056
|
-
/**
|
|
5057
|
-
* Run the protocol: one call, one bounded repair turn on a structurally
|
|
5058
|
-
* malformed reply, then decode. Zero valid rows from a well-formed reply is an
|
|
5059
|
-
* honest null, not a failure.
|
|
5060
|
-
*/
|
|
5061
|
-
async function runPrimeExchange(options) {
|
|
5062
|
-
const { contract } = options;
|
|
5063
|
-
const turns = [];
|
|
5064
|
-
const repair = {
|
|
5065
|
-
attempted: false,
|
|
5066
|
-
succeeded: null
|
|
5067
|
-
};
|
|
5068
|
-
const first = await callPrimeTurn(options, options.prompt);
|
|
5069
|
-
if (!first.ok) return {
|
|
5070
|
-
ok: false,
|
|
5071
|
-
failure: first.failure,
|
|
5072
|
-
usage: mergeTurns(turns),
|
|
5073
|
-
turns,
|
|
5074
|
-
repair
|
|
5075
|
-
};
|
|
5076
|
-
turns.push({
|
|
5077
|
-
turn: "first",
|
|
5078
|
-
usage: first.usage,
|
|
5079
|
-
rawUsage: first.rawUsage
|
|
5080
|
-
});
|
|
5081
|
-
let reply = first.content;
|
|
5082
|
-
let parsed = extractPrimeJsonObject(reply);
|
|
5083
|
-
let defect = primeReplyDefect(parsed, contract.rowsField);
|
|
5084
|
-
if (defect !== null && options.repair) {
|
|
5085
|
-
repair.attempted = true;
|
|
5086
|
-
const second = await callPrimeTurn(options, buildPrimeRepairPrompt({
|
|
5087
|
-
defect,
|
|
5088
|
-
previousReply: reply,
|
|
5089
|
-
repairContractLines: contract.repairContractLines
|
|
5090
|
-
}));
|
|
5091
|
-
if (!second.ok) return {
|
|
5092
|
-
ok: false,
|
|
5093
|
-
failure: second.failure,
|
|
5094
|
-
usage: mergeTurns(turns),
|
|
5095
|
-
turns,
|
|
5096
|
-
repair,
|
|
5097
|
-
reply
|
|
5098
|
-
};
|
|
5099
|
-
turns.push({
|
|
5100
|
-
turn: "repair",
|
|
5101
|
-
usage: second.usage,
|
|
5102
|
-
rawUsage: second.rawUsage
|
|
5103
|
-
});
|
|
5104
|
-
reply = second.content;
|
|
5105
|
-
parsed = extractPrimeJsonObject(reply);
|
|
5106
|
-
defect = primeReplyDefect(parsed, contract.rowsField);
|
|
5107
|
-
repair.succeeded = defect === null;
|
|
5108
|
-
}
|
|
5109
|
-
const usage = mergeTurns(turns);
|
|
5110
|
-
if (defect !== null) return {
|
|
5111
|
-
ok: false,
|
|
5112
|
-
failure: {
|
|
5113
|
-
kind: "malformed-reply",
|
|
5114
|
-
message: `${defect} in prime reply${repair.attempted ? " even after the bounded repair turn" : ""}`
|
|
5115
|
-
},
|
|
5116
|
-
usage,
|
|
5117
|
-
turns,
|
|
5118
|
-
repair,
|
|
5119
|
-
reply
|
|
5120
|
-
};
|
|
5121
|
-
const rawRows = parsed[contract.rowsField];
|
|
5122
|
-
const rows = [];
|
|
5123
|
-
const rejected = [];
|
|
5124
|
-
let overflow = 0;
|
|
5125
|
-
rawRows.forEach((row, index) => {
|
|
5126
|
-
const decoded = contract.decodeRow(row, index);
|
|
5127
|
-
if (!decoded.ok) {
|
|
5128
|
-
rejected.push({
|
|
5129
|
-
index,
|
|
5130
|
-
reason: decoded.reason
|
|
5131
|
-
});
|
|
5132
|
-
return;
|
|
5133
|
-
}
|
|
5134
|
-
if (contract.maxRows !== void 0 && rows.length >= contract.maxRows) {
|
|
5135
|
-
overflow += 1;
|
|
5136
|
-
return;
|
|
5137
|
-
}
|
|
5138
|
-
rows.push(decoded.row);
|
|
5139
|
-
});
|
|
5140
|
-
const answer = parsed.answer;
|
|
5141
|
-
return {
|
|
5142
|
-
ok: true,
|
|
5143
|
-
answer: typeof answer === "string" ? answer : null,
|
|
5144
|
-
rows,
|
|
5145
|
-
rejected,
|
|
5146
|
-
reportedRows: rawRows.length,
|
|
5147
|
-
overflow,
|
|
5148
|
-
usage,
|
|
5149
|
-
turns,
|
|
5150
|
-
repair,
|
|
5151
|
-
reply
|
|
5152
|
-
};
|
|
5153
|
-
}
|
|
5154
|
-
async function callPrimeTurn(options, content) {
|
|
5155
|
-
const { transport, url, model, timeoutMs, signal } = options;
|
|
5156
|
-
const controller = new AbortController();
|
|
5157
|
-
const forwardAbort = () => controller.abort(signal?.reason);
|
|
5158
|
-
if (signal?.aborted) controller.abort(signal.reason);
|
|
5159
|
-
else signal?.addEventListener("abort", forwardAbort, { once: true });
|
|
5160
|
-
const deadline = setTimeout(() => controller.abort(), timeoutMs);
|
|
5161
|
-
let result;
|
|
5162
|
-
try {
|
|
5163
|
-
result = await transport({
|
|
5164
|
-
url,
|
|
5165
|
-
body: {
|
|
5166
|
-
model,
|
|
5167
|
-
messages: [{
|
|
5168
|
-
role: "user",
|
|
5169
|
-
content
|
|
5170
|
-
}]
|
|
5171
|
-
},
|
|
5172
|
-
signal: controller.signal
|
|
5173
|
-
});
|
|
5174
|
-
} catch (error) {
|
|
5175
|
-
if (signal?.aborted) return {
|
|
5176
|
-
ok: false,
|
|
5177
|
-
failure: {
|
|
5178
|
-
kind: "aborted",
|
|
5179
|
-
message: "prime exchange cancelled by the caller",
|
|
5180
|
-
cause: error
|
|
5181
|
-
}
|
|
5182
|
-
};
|
|
5183
|
-
if (controller.signal.aborted) return {
|
|
5184
|
-
ok: false,
|
|
5185
|
-
failure: {
|
|
5186
|
-
kind: "deadline",
|
|
5187
|
-
message: `bridge call exceeded ${timeoutMs}ms`
|
|
5188
|
-
}
|
|
5189
|
-
};
|
|
5190
|
-
return {
|
|
5191
|
-
ok: false,
|
|
5192
|
-
failure: {
|
|
5193
|
-
kind: "transport",
|
|
5194
|
-
message: `bridge transport failure: ${error instanceof Error ? error.message : String(error)}`
|
|
5195
|
-
}
|
|
5196
|
-
};
|
|
5197
|
-
} finally {
|
|
5198
|
-
clearTimeout(deadline);
|
|
5199
|
-
signal?.removeEventListener("abort", forwardAbort);
|
|
5200
|
-
}
|
|
5201
|
-
if (result.status !== 200) {
|
|
5202
|
-
const bodySnippet = result.text.slice(0, 500);
|
|
5203
|
-
return {
|
|
5204
|
-
ok: false,
|
|
5205
|
-
failure: {
|
|
5206
|
-
kind: "http-status",
|
|
5207
|
-
message: `bridge HTTP ${result.status}: ${bodySnippet}`,
|
|
5208
|
-
status: result.status,
|
|
5209
|
-
bodySnippet
|
|
5210
|
-
}
|
|
5211
|
-
};
|
|
5212
|
-
}
|
|
5213
|
-
let response;
|
|
5214
|
-
try {
|
|
5215
|
-
response = JSON.parse(result.text);
|
|
5216
|
-
} catch {
|
|
5217
|
-
return {
|
|
5218
|
-
ok: false,
|
|
5219
|
-
failure: {
|
|
5220
|
-
kind: "unparseable-json",
|
|
5221
|
-
message: `bridge returned unparseable JSON (${result.text.length} bytes)`
|
|
5222
|
-
}
|
|
5223
|
-
};
|
|
5224
|
-
}
|
|
5225
|
-
const replyContent = primeReplyContent(response);
|
|
5226
|
-
if (replyContent === null) return {
|
|
5227
|
-
ok: false,
|
|
5228
|
-
failure: {
|
|
5229
|
-
kind: "no-content",
|
|
5230
|
-
message: "bridge reply carries no message content"
|
|
5231
|
-
}
|
|
5232
|
-
};
|
|
5233
|
-
const rawUsage = primeReplyUsage(response);
|
|
5234
|
-
return {
|
|
5235
|
-
ok: true,
|
|
5236
|
-
content: replyContent,
|
|
5237
|
-
usage: normalizePrimeUsage(rawUsage),
|
|
5238
|
-
rawUsage
|
|
5239
|
-
};
|
|
5240
|
-
}
|
|
5241
|
-
/**
|
|
5242
|
-
* Fold from the FIRST turn, never from an empty receipt: an all-null identity
|
|
5243
|
-
* would poison every side it merged with and erase counts the bridge reported.
|
|
5244
|
-
*/
|
|
5245
|
-
function mergeTurns(turns) {
|
|
5246
|
-
if (turns.length === 0) return emptyPrimeRawUsage();
|
|
5247
|
-
return turns.slice(1).reduce((total, turn) => mergePrimeRawUsage(total, turn.usage), turns[0].usage);
|
|
5248
|
-
}
|
|
5249
|
-
function primeReplyContent(response) {
|
|
5250
|
-
if (typeof response !== "object" || response === null) return null;
|
|
5251
|
-
const choices = response.choices;
|
|
5252
|
-
if (!Array.isArray(choices) || choices.length === 0) return null;
|
|
5253
|
-
const message = choices[0]?.message;
|
|
5254
|
-
if (typeof message !== "object" || message === null) return null;
|
|
5255
|
-
const content = message.content;
|
|
5256
|
-
return typeof content === "string" && content.length > 0 ? content : null;
|
|
5257
|
-
}
|
|
5258
|
-
function primeReplyUsage(response) {
|
|
5259
|
-
if (typeof response !== "object" || response === null) return null;
|
|
5260
|
-
return response.usage ?? null;
|
|
5261
|
-
}
|
|
5262
|
-
/**
|
|
5263
|
-
* Render, measure, fall back to the capped projection, re-measure, fail loud.
|
|
5264
|
-
*
|
|
5265
|
-
* Inline is the only delivery prime has, so an oversized trajectory is a
|
|
5266
|
-
* refusal rather than a silent truncation: dropping spans would understate the
|
|
5267
|
-
* trajectory and the analyst would answer a question about a different run.
|
|
5268
|
-
*/
|
|
5269
|
-
async function projectPrimeTrajectory(source, limits) {
|
|
5270
|
-
let fetch = "full";
|
|
5271
|
-
let items = await source.full();
|
|
5272
|
-
if (items === null) {
|
|
5273
|
-
fetch = "capped";
|
|
5274
|
-
items = await source.capped();
|
|
5275
|
-
}
|
|
5276
|
-
let rendered = JSON.stringify(items);
|
|
5277
|
-
if (rendered.length > limits.maxInlineChars && fetch === "full") {
|
|
5278
|
-
fetch = "capped";
|
|
5279
|
-
items = await source.capped();
|
|
5280
|
-
rendered = JSON.stringify(items);
|
|
5281
|
-
}
|
|
5282
|
-
if (rendered.length > limits.maxInlineChars) return {
|
|
5283
|
-
ok: false,
|
|
5284
|
-
reason: `trajectory renders to ${rendered.length} chars even at ${source.cappedDescription}; inline delivery impossible`,
|
|
5285
|
-
renderedChars: rendered.length
|
|
5286
|
-
};
|
|
5287
|
-
return {
|
|
5288
|
-
ok: true,
|
|
5289
|
-
items,
|
|
5290
|
-
rendered,
|
|
5291
|
-
delivery: {
|
|
5292
|
-
mode: "inline-json",
|
|
5293
|
-
fetch,
|
|
5294
|
-
renderedChars: rendered.length
|
|
5295
|
-
}
|
|
5296
|
-
};
|
|
5297
|
-
}
|
|
5298
|
-
/**
|
|
5299
|
-
* Digest of everything a consumer can send to the bridge under the prime
|
|
5300
|
-
* protocol, recorded per observation so a prime result names the exact contract
|
|
5301
|
-
* that produced it.
|
|
5302
|
-
*
|
|
5303
|
-
* Computed over the ACTUALLY composed contract, so two consumers that both
|
|
5304
|
-
* stamp `analyst_id: 'prime'` while asking materially different questions get
|
|
5305
|
-
* different digests by construction. That is what makes 'prime' a reproducible
|
|
5306
|
-
* claim rather than a label.
|
|
5307
|
-
*/
|
|
5308
|
-
function primeProtocolSha256(identity) {
|
|
5309
|
-
return createHash("sha256").update(JSON.stringify({
|
|
5310
|
-
kind: "prime-analyst-protocol",
|
|
5311
|
-
question: identity.question,
|
|
5312
|
-
taskPrompt: identity.taskDefinition ?? null,
|
|
5313
|
-
outputContract: identity.contractLines,
|
|
5314
|
-
repairContract: buildPrimeRepairPrompt({
|
|
5315
|
-
defect: "<defect>",
|
|
5316
|
-
previousReply: "<previous-reply>",
|
|
5317
|
-
repairContractLines: identity.repairContractLines
|
|
5318
|
-
}),
|
|
5319
|
-
limits: identity.limits
|
|
5320
|
-
})).digest("hex");
|
|
5321
|
-
}
|
|
5322
|
-
//#endregion
|
|
5323
5304
|
//#region src/analyst/benchmark-runner-prime.ts
|
|
5324
5305
|
/**
|
|
5325
5306
|
* Prime analyst arm: the RLM coding agent reached through an OpenAI-compatible
|
|
5326
5307
|
* cli-bridge solves the CodeTraceBench incorrect-step task as a one-shot trace
|
|
5327
5308
|
* analyst.
|
|
5328
5309
|
*
|
|
5329
|
-
* The
|
|
5330
|
-
*
|
|
5331
|
-
*
|
|
5332
|
-
*
|
|
5333
|
-
*
|
|
5310
|
+
* The arm is expressed as an `AnalystDefinition`
|
|
5311
|
+
* (`primeCodeTraceAnalystDefinition`): the question, task text, output
|
|
5312
|
+
* contract, inline projection budget, and repair-turn declaration are all
|
|
5313
|
+
* definition content, and `createPrimeBenchmarkRunner` is a thin shell that
|
|
5314
|
+
* builds the definition and runs it through the inline strategy below. The
|
|
5315
|
+
* same strategy is what `bindAnalyst` (./bind) dispatches to, so a compiled
|
|
5316
|
+
* definition and this entry point send byte-identical requests — the parity
|
|
5317
|
+
* suite asserts exactly that.
|
|
5334
5318
|
*
|
|
5335
|
-
* The protocol
|
|
5319
|
+
* The protocol machinery — prompt composition, the bounded repair turn, reply
|
|
5336
5320
|
* extraction, the projection ladder, usage normalization — lives in
|
|
5337
|
-
* `./prime-protocol`, which knows nothing about CodeTraceBench. This file
|
|
5338
|
-
* the benchmark's binding to it
|
|
5339
|
-
*
|
|
5321
|
+
* `./prime-protocol`, which knows nothing about CodeTraceBench. This file adds
|
|
5322
|
+
* the benchmark's binding to it (block row grammar, store-backed projection,
|
|
5323
|
+
* observation shape) plus the projection-generic inline execution strategy.
|
|
5340
5324
|
*
|
|
5341
5325
|
* Trajectory delivery is inline JSON in the prompt. The dspy typed path binds
|
|
5342
5326
|
* the viewTrace span projection as a REPL variable; prime has no REPL, so the
|
|
5343
5327
|
* same projection is serialized into the prompt. When the full projection is
|
|
5344
|
-
* oversized the
|
|
5328
|
+
* oversized the strategy falls back to chunked viewSpans over the same
|
|
5345
5329
|
* projection surface with a per-attribute byte cap, and fails loud if the
|
|
5346
5330
|
* result still exceeds the inline budget.
|
|
5347
5331
|
*
|
|
@@ -5443,56 +5427,142 @@ const PRIME_BLOCK_CONTRACT = {
|
|
|
5443
5427
|
}
|
|
5444
5428
|
};
|
|
5445
5429
|
/**
|
|
5446
|
-
* Digest of everything this
|
|
5430
|
+
* Digest of everything this arm can send to the bridge, recorded per
|
|
5447
5431
|
* observation so a prime result names the exact contract that produced it.
|
|
5448
5432
|
*/
|
|
5449
5433
|
function primeAnalystProtocolSha256() {
|
|
5450
5434
|
return primeProtocolSha256(PRIME_PROTOCOL_IDENTITY);
|
|
5451
5435
|
}
|
|
5452
|
-
/**
|
|
5436
|
+
/**
|
|
5437
|
+
* The prime arm as a declarative unit. CodeTraceBench-only: the question,
|
|
5438
|
+
* task text, and block grammar speak its incorrect-step definition.
|
|
5439
|
+
*/
|
|
5440
|
+
function primeCodeTraceAnalystDefinition(args) {
|
|
5441
|
+
return {
|
|
5442
|
+
id: PRIME_ANALYST_ID,
|
|
5443
|
+
description: "One-shot RLM over an OpenAI-compatible bridge answering the CodeTraceBench incorrect-step task.",
|
|
5444
|
+
version: "1.0.0",
|
|
5445
|
+
area: "incorrect",
|
|
5446
|
+
profile: {},
|
|
5447
|
+
question: PRIME_QUESTION,
|
|
5448
|
+
taskDefinition: CODE_TRACE_BENCH_ANALYST_PROMPT,
|
|
5449
|
+
projection: {
|
|
5450
|
+
mode: "inline",
|
|
5451
|
+
maxInlineChars: MAX_INLINE_TRAJECTORY_CHARS,
|
|
5452
|
+
cappedAttributeBytes: CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP
|
|
5453
|
+
},
|
|
5454
|
+
replyContract: PRIME_BLOCK_CONTRACT,
|
|
5455
|
+
contractLimits: {
|
|
5456
|
+
maxBlocks: 16,
|
|
5457
|
+
maxBlockSteps: 12
|
|
5458
|
+
},
|
|
5459
|
+
budget: { timeoutMs: args.timeoutMs },
|
|
5460
|
+
repair: { turns: args.repairTurns },
|
|
5461
|
+
protocolSha256: primeAnalystProtocolSha256(),
|
|
5462
|
+
binding: {
|
|
5463
|
+
kind: "inline",
|
|
5464
|
+
subjectFromCaseId: trajectoryIdFromCaseId,
|
|
5465
|
+
baseMetadata: {
|
|
5466
|
+
analysisMode: "prime-rlm",
|
|
5467
|
+
engine: "prime"
|
|
5468
|
+
},
|
|
5469
|
+
header(subject, spans) {
|
|
5470
|
+
const stepSpans = spans.filter((span) => /^step-\d+$/.test(String(span.span_id)));
|
|
5471
|
+
if (stepSpans.length === 0) throw new PrimeTraceProjectionError(`no step-<n> spans in trace '${subject}'`);
|
|
5472
|
+
return `TRAJECTORY (trace_id ${subject}; ${stepSpans.length} assistant step spans; full span projection as JSON):`;
|
|
5473
|
+
},
|
|
5474
|
+
trailer(_subject, spans) {
|
|
5475
|
+
const finalVerification = spans.filter(isFinalVerificationSpan);
|
|
5476
|
+
return finalVerification.length > 0 ? `FINAL VERIFICATION SPANS:\n${JSON.stringify(finalVerification)}` : "FINAL VERIFICATION: unavailable for this trajectory — trace backward from the latest failure evidence inside the trajectory itself.";
|
|
5477
|
+
},
|
|
5478
|
+
async expandRows({ subject, rows, store, analystId, signal }) {
|
|
5479
|
+
const expanded = await expandCodeTraceFailureBlocks({
|
|
5480
|
+
trajectoryId: subject,
|
|
5481
|
+
blocks: rows,
|
|
5482
|
+
store,
|
|
5483
|
+
analystId,
|
|
5484
|
+
...signal ? { signal } : {}
|
|
5485
|
+
});
|
|
5486
|
+
return {
|
|
5487
|
+
findings: expanded.findings,
|
|
5488
|
+
diagnostics: expanded.diagnostics
|
|
5489
|
+
};
|
|
5490
|
+
}
|
|
5491
|
+
}
|
|
5492
|
+
};
|
|
5493
|
+
}
|
|
5494
|
+
/** Thin shell: validate options, declare the definition, run the inline strategy. */
|
|
5453
5495
|
function createPrimeBenchmarkRunner(options) {
|
|
5454
|
-
const baseUrl = requiredString(options.baseUrl, "baseUrl").replace(/\/+$/, "");
|
|
5455
|
-
const model = requiredString(options.model, "model");
|
|
5456
5496
|
const timeoutMs = positiveSafeInteger(options.timeoutMs, "timeoutMs");
|
|
5457
5497
|
const repair = options.repair;
|
|
5458
5498
|
if (typeof repair !== "boolean") throw new TypeError("repair must be a boolean");
|
|
5459
|
-
|
|
5460
|
-
|
|
5499
|
+
return runInlineAnalystDefinition(primeCodeTraceAnalystDefinition({
|
|
5500
|
+
timeoutMs,
|
|
5501
|
+
repairTurns: repair ? 1 : 0
|
|
5502
|
+
}), {
|
|
5503
|
+
baseUrl: options.baseUrl,
|
|
5504
|
+
model: options.model,
|
|
5505
|
+
...options.transport ? { transport: options.transport } : {},
|
|
5506
|
+
...options.pricing ? { pricing: options.pricing } : {}
|
|
5507
|
+
});
|
|
5508
|
+
}
|
|
5509
|
+
/**
|
|
5510
|
+
* Compile an inline-projection definition into a runnable arm. Projection,
|
|
5511
|
+
* prompt composition, the bounded repair turn, and usage accounting are all
|
|
5512
|
+
* driven by the definition; nothing in this strategy names a benchmark.
|
|
5513
|
+
*/
|
|
5514
|
+
function runInlineAnalystDefinition(definition, transports) {
|
|
5515
|
+
const { projection, binding } = definition;
|
|
5516
|
+
if (projection.mode !== "inline" || binding.kind !== "inline") throw new AnalystExpressivenessError(`the inline strategy compiles only inline projections; definition '${definition.id}' declares projection '${projection.mode}' with a '${binding.kind}' binding`);
|
|
5517
|
+
if (definition.repair.turns > 1) throw new AnalystExpressivenessError(`the inline exchange grants at most one bounded repair turn; definition '${definition.id}' declares ${definition.repair.turns}`);
|
|
5518
|
+
const baseUrl = requiredString(transports.baseUrl, "baseUrl").replace(/\/+$/, "");
|
|
5519
|
+
const model = requiredString(transports.model, "model");
|
|
5520
|
+
const timeoutMs = positiveSafeInteger(definition.budget.timeoutMs, "timeoutMs");
|
|
5521
|
+
const repair = definition.repair.turns === 1;
|
|
5522
|
+
const pricing = transports.pricing ?? pricingForModel(model);
|
|
5523
|
+
const transport = transports.transport ?? nodeHttpPrimeBridgeTransport();
|
|
5461
5524
|
const url = `${baseUrl}/v1/chat/completions`;
|
|
5462
5525
|
return {
|
|
5463
|
-
id:
|
|
5526
|
+
id: definition.id,
|
|
5464
5527
|
async analyze(input, context) {
|
|
5465
|
-
const
|
|
5528
|
+
const subject = binding.subjectFromCaseId(context.caseId);
|
|
5466
5529
|
let usage;
|
|
5467
5530
|
let metadata = {
|
|
5468
|
-
|
|
5469
|
-
engine: "prime",
|
|
5531
|
+
...binding.baseMetadata,
|
|
5470
5532
|
bridgeUrl: baseUrl,
|
|
5471
5533
|
model,
|
|
5472
|
-
protocolSha256:
|
|
5534
|
+
protocolSha256: definition.protocolSha256
|
|
5473
5535
|
};
|
|
5474
5536
|
try {
|
|
5475
5537
|
const store = input.traceStore;
|
|
5476
|
-
if (!store) throw new Error(
|
|
5477
|
-
const
|
|
5478
|
-
|
|
5538
|
+
if (!store) throw new Error(`inline analyst '${definition.id}' requires a trace store`);
|
|
5539
|
+
const storeContext = context.signal ? { signal: context.signal } : void 0;
|
|
5540
|
+
const projected = await projectPrimeTrajectory(inlineProjectionSource(store, subject, projection.cappedAttributeBytes, storeContext), { maxInlineChars: projection.maxInlineChars });
|
|
5541
|
+
if (!projected.ok) throw new PrimeTraceProjectionError(projected.reason);
|
|
5479
5542
|
const delivery = {
|
|
5480
|
-
mode:
|
|
5481
|
-
fetch:
|
|
5482
|
-
perAttributeByteCap:
|
|
5483
|
-
renderedChars:
|
|
5543
|
+
mode: projected.delivery.mode,
|
|
5544
|
+
fetch: projected.delivery.fetch === "full" ? "view-trace" : "view-spans-chunked",
|
|
5545
|
+
perAttributeByteCap: projected.delivery.fetch === "full" ? null : projection.cappedAttributeBytes,
|
|
5546
|
+
renderedChars: projected.delivery.renderedChars
|
|
5484
5547
|
};
|
|
5485
5548
|
metadata = {
|
|
5486
5549
|
...metadata,
|
|
5487
5550
|
delivery
|
|
5488
5551
|
};
|
|
5489
|
-
const prompt =
|
|
5552
|
+
const prompt = buildPrimePrompt({
|
|
5553
|
+
question: definition.question,
|
|
5554
|
+
...definition.taskDefinition === void 0 ? {} : { taskDefinition: definition.taskDefinition },
|
|
5555
|
+
contractLines: definition.replyContract.contractLines,
|
|
5556
|
+
trajectoryHeader: binding.header(subject, projected.items),
|
|
5557
|
+
renderedTrajectory: projected.rendered,
|
|
5558
|
+
trailer: binding.trailer(subject, projected.items)
|
|
5559
|
+
});
|
|
5490
5560
|
metadata = {
|
|
5491
5561
|
...metadata,
|
|
5492
5562
|
promptChars: prompt.length
|
|
5493
5563
|
};
|
|
5494
5564
|
const outcome = await runPrimeExchange({
|
|
5495
|
-
contract:
|
|
5565
|
+
contract: definition.replyContract,
|
|
5496
5566
|
prompt,
|
|
5497
5567
|
transport,
|
|
5498
5568
|
url,
|
|
@@ -5520,11 +5590,11 @@ function createPrimeBenchmarkRunner(options) {
|
|
|
5520
5590
|
};
|
|
5521
5591
|
throw primeFailureError(outcome.failure);
|
|
5522
5592
|
}
|
|
5523
|
-
const expanded = await
|
|
5524
|
-
|
|
5525
|
-
|
|
5593
|
+
const expanded = await binding.expandRows({
|
|
5594
|
+
subject,
|
|
5595
|
+
rows: outcome.rows,
|
|
5526
5596
|
store,
|
|
5527
|
-
analystId:
|
|
5597
|
+
analystId: definition.id,
|
|
5528
5598
|
...context.signal ? { signal: context.signal } : {}
|
|
5529
5599
|
});
|
|
5530
5600
|
return {
|
|
@@ -5555,13 +5625,13 @@ function createPrimeBenchmarkRunner(options) {
|
|
|
5555
5625
|
* the full viewTrace projection, or the chunked viewSpans projection at a
|
|
5556
5626
|
* per-attribute byte cap.
|
|
5557
5627
|
*/
|
|
5558
|
-
function
|
|
5628
|
+
function inlineProjectionSource(store, trajectoryId, cappedAttributeBytes, context) {
|
|
5559
5629
|
return {
|
|
5560
5630
|
async full() {
|
|
5561
5631
|
return (await store.viewTrace({ trace_id: trajectoryId }, context)).spans ?? null;
|
|
5562
5632
|
},
|
|
5563
|
-
capped: () => projectSpansChunked(store, trajectoryId, context),
|
|
5564
|
-
cappedDescription: `per-attribute cap ${
|
|
5633
|
+
capped: () => projectSpansChunked(store, trajectoryId, cappedAttributeBytes, context),
|
|
5634
|
+
cappedDescription: `per-attribute cap ${cappedAttributeBytes}`
|
|
5565
5635
|
};
|
|
5566
5636
|
}
|
|
5567
5637
|
/**
|
|
@@ -5570,7 +5640,7 @@ function codeTraceProjectionSource(store, trajectoryId, context) {
|
|
|
5570
5640
|
* id must project or the case fails loud — a silently dropped span would
|
|
5571
5641
|
* understate the trajectory.
|
|
5572
5642
|
*/
|
|
5573
|
-
async function projectSpansChunked(store, trajectoryId, context) {
|
|
5643
|
+
async function projectSpansChunked(store, trajectoryId, cappedAttributeBytes, context) {
|
|
5574
5644
|
const enumeration = await store.viewTrace({
|
|
5575
5645
|
trace_id: trajectoryId,
|
|
5576
5646
|
per_attribute_byte_cap: SPAN_ID_ENUMERATION_ATTRIBUTE_BYTE_CAP
|
|
@@ -5589,26 +5659,13 @@ async function projectSpansChunked(store, trajectoryId, context) {
|
|
|
5589
5659
|
const result = await store.viewSpans({
|
|
5590
5660
|
trace_id: trajectoryId,
|
|
5591
5661
|
span_ids: chunk,
|
|
5592
|
-
per_attribute_byte_cap:
|
|
5662
|
+
per_attribute_byte_cap: cappedAttributeBytes
|
|
5593
5663
|
}, context);
|
|
5594
5664
|
if (result.missing_span_ids.length > 0 || result.omitted_span_ids.length > 0 || result.spans.length !== chunk.length) throw new PrimeTraceProjectionError(`viewSpans projected ${result.spans.length}/${chunk.length} spans for chunk at ${index} of '${trajectoryId}'`);
|
|
5595
5665
|
projected.push(...result.spans);
|
|
5596
5666
|
}
|
|
5597
5667
|
return projected;
|
|
5598
5668
|
}
|
|
5599
|
-
function buildCodeTracePrompt(trajectoryId, spans, renderedSpans) {
|
|
5600
|
-
const stepSpans = spans.filter((span) => /^step-\d+$/.test(String(span.span_id)));
|
|
5601
|
-
if (stepSpans.length === 0) throw new PrimeTraceProjectionError(`no step-<n> spans in trace '${trajectoryId}'`);
|
|
5602
|
-
const finalVerification = spans.filter(isFinalVerificationSpan);
|
|
5603
|
-
return buildPrimePrompt({
|
|
5604
|
-
question: PRIME_QUESTION,
|
|
5605
|
-
taskDefinition: CODE_TRACE_BENCH_ANALYST_PROMPT,
|
|
5606
|
-
contractLines: PRIME_OUTPUT_CONTRACT_LINES,
|
|
5607
|
-
trajectoryHeader: `TRAJECTORY (trace_id ${trajectoryId}; ${stepSpans.length} assistant step spans; full span projection as JSON):`,
|
|
5608
|
-
renderedTrajectory: renderedSpans,
|
|
5609
|
-
trailer: finalVerification.length > 0 ? `FINAL VERIFICATION SPANS:\n${JSON.stringify(finalVerification)}` : "FINAL VERIFICATION: unavailable for this trajectory — trace backward from the latest failure evidence inside the trajectory itself."
|
|
5610
|
-
});
|
|
5611
|
-
}
|
|
5612
5669
|
/** Map the protocol's terminal reason onto this benchmark's typed error classes. */
|
|
5613
5670
|
function primeFailureError(failure) {
|
|
5614
5671
|
switch (failure.kind) {
|
|
@@ -6353,6 +6410,6 @@ function shellQuote(value) {
|
|
|
6353
6410
|
return /^[a-zA-Z0-9_./:@%+=,-]+$/.test(value) ? value : `'${value.replaceAll("'", "'\\''")}'`;
|
|
6354
6411
|
}
|
|
6355
6412
|
//#endregion
|
|
6356
|
-
export { ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as $, summarizeCodeTraceCalibration as A, publicBenchmarkSystemPrompt as B,
|
|
6413
|
+
export { ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as $, summarizeCodeTraceCalibration as A, publicBenchmarkSystemPrompt as B, analystDefinitionAsymmetries as C, expandCodeTraceFailureBlocks as D, emptyPublicBenchmarkRunner as E, CODE_TRACE_BENCH_ANALYST_PROMPT as F, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as G, appendVerificationArtifactsToOtlp as H, MAX_INCORRECT_BLOCKS as I, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as J, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as K, MAX_INCORRECT_BLOCK_STEPS as L, analystInstructionsOverrideFromText as M, effectiveAnalystProtocolSha256 as N, readAnalystBenchmarkArtifact as O, readAnalystInstructionsOverride as P, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as Q, publicBenchmarkProtocolSha256 as R, AnalystExpressivenessError as S, adaptPublicBenchmarkFindings as T, loadCodeTraceVerificationArtifacts as U, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as V, parseVerificationOutcome as W, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as X, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as Y, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as Z, runReplVariableAnalystDefinition as _, primeAnalystProtocolSha256 as a, ANALYST_BENCHMARK_OBSERVATIONS_FILE as at, runChunkedAnalystDefinition as b, nodeHttpPrimeBridgeTransport as c, summarizeAgentRxCalibration as ct, publicBenchmarkDistributions as d, agentRxBenchmarkCase as dt, analystBenchmarkDependencyLockDigest as et, publicBenchmarkSelectionReport as f, agentRxPredictionsToFindings as ft, rlmEngineLimits as g, publicRlmAnalystDefinition as h, normalizeBenchmarkLabel as ht, createPrimeBenchmarkRunner as i, ANALYST_BENCHMARK_MANIFEST_FILE as it, compareAnalystRunners as j, renderCodeTraceCalibrationMarkdown as k, loadPublicBenchmarkRows as l, codeTraceBenchCase as lt, createPublicBenchmarkRlmRunner as m, roundAgentRxStep as mt, runAnalystBenchmarkCommand as n, ANALYST_BENCHMARK_COST_LEDGER_FILE as nt, primeCodeTraceAnalystDefinition as o, AGENT_RX_UPSTREAM_REVISION as ot, selectPublicBenchmarkRows as p, normalizeAgentRxCategory as pt, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as q, renderAnalystBenchmarkMarkdown as r, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as rt, runInlineAnalystDefinition as s, renderAgentRxCalibrationMarkdown as st, ANALYST_BENCHMARK_HELP as t, analystBenchmarkImplementationDigest as tt, preparePublicAnalystBenchmark as u, codeTracerPredictionsToFindings as ut, createPublicBenchmarkDirectRunner as v, analystDefinitionProtocolSha256 as w, decodeReplyRows as x, publicDirectAnalystDefinition as y, publicBenchmarkRlmInstructions as z };
|
|
6357
6414
|
|
|
6358
|
-
//# sourceMappingURL=benchmark-command-
|
|
6415
|
+
//# sourceMappingURL=benchmark-command-BKENp2s5.js.map
|