@tangle-network/agent-eval 0.144.4 → 0.144.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +366 -19
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +8 -8
- package/dist/{benchmark-CYtcIF2V.js → benchmark-B181aMF9.js} +2 -2
- package/dist/{benchmark-CYtcIF2V.js.map → benchmark-B181aMF9.js.map} +1 -1
- package/dist/{benchmark-CP6kWfj8.d.ts → benchmark-J9Qe6j2_.d.ts} +3 -3
- package/dist/{benchmark-CP6kWfj8.d.ts.map → benchmark-J9Qe6j2_.d.ts.map} +1 -1
- package/dist/{benchmark-command-9FTgq6Fg.js → benchmark-command-CQd78YHt.js} +922 -36
- package/dist/benchmark-command-CQd78YHt.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CkG1bWFa.js → benchmarks-BEOkuvIg.js} +3 -3
- package/dist/{benchmarks-CkG1bWFa.js.map → benchmarks-BEOkuvIg.js.map} +1 -1
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +3 -3
- package/dist/{campaign-DjGFyPxH.js → campaign-CXsdyym7.js} +4 -4
- package/dist/{campaign-DjGFyPxH.js.map → campaign-CXsdyym7.js.map} +1 -1
- package/dist/cli.js +2 -2
- package/dist/{client-Bbht4xxl.d.ts → client-0JI64ovJ.d.ts} +4 -4
- package/dist/{client-Bbht4xxl.d.ts.map → client-0JI64ovJ.d.ts.map} +1 -1
- package/dist/{completion-verifier-EJERfFwF.d.ts → completion-verifier-CBiee74w.d.ts} +5 -5
- package/dist/{completion-verifier-EJERfFwF.d.ts.map → completion-verifier-CBiee74w.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +10 -10
- package/dist/contract/index.js +5 -5
- package/dist/control.d.ts +2 -2
- package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
- package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
- package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
- package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
- package/dist/{default-registry-SOyHB6qG.js → default-registry-Dta70shL.js} +4 -4
- package/dist/{default-registry-SOyHB6qG.js.map → default-registry-Dta70shL.js.map} +1 -1
- package/dist/{default-registry-iXfu2trt.d.ts → default-registry-J9m-_tya.d.ts} +5 -5
- package/dist/{default-registry-iXfu2trt.d.ts.map → default-registry-J9m-_tya.d.ts.map} +1 -1
- package/dist/{dspy-rlm-engine-IRCG8kdi.js → dspy-rlm-engine-19FQEMBK.js} +2 -2
- package/dist/{dspy-rlm-engine-IRCG8kdi.js.map → dspy-rlm-engine-19FQEMBK.js.map} +1 -1
- package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
- package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
- package/dist/errors-D-LKuDhb.js.map +1 -1
- package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-CfLQQs9B.js} +2 -2
- package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-CfLQQs9B.js.map} +1 -1
- package/dist/{exact-types-BygCBR4L.d.ts → exact-types-CBYF5MGd.d.ts} +2 -2
- package/dist/{exact-types-BygCBR4L.d.ts.map → exact-types-CBYF5MGd.d.ts.map} +1 -1
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-Q9c0L3eY.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-Q9c0L3eY.d.ts.map} +1 -1
- package/dist/{extract-usage-7l1Xq5ti.js → extract-usage-BW27f3XW.js} +2 -2
- package/dist/{extract-usage-7l1Xq5ti.js.map → extract-usage-BW27f3XW.js.map} +1 -1
- package/dist/{feedback-trajectory-CSIkRLQX.d.ts → feedback-trajectory-GgoS0-MK.d.ts} +4 -4
- package/dist/{feedback-trajectory-CSIkRLQX.d.ts.map → feedback-trajectory-GgoS0-MK.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-D_qTihaQ.d.ts → index-4XwggC10.d.ts} +5 -5
- package/dist/{index-D_qTihaQ.d.ts.map → index-4XwggC10.d.ts.map} +1 -1
- package/dist/{index-CNOCxBLh.d.ts → index-B6-B0zTB.d.ts} +2 -2
- package/dist/{index-CNOCxBLh.d.ts.map → index-B6-B0zTB.d.ts.map} +1 -1
- package/dist/{index-BuHs_OnD.d.ts → index-BIL5vxxt.d.ts} +12 -12
- package/dist/{index-BuHs_OnD.d.ts.map → index-BIL5vxxt.d.ts.map} +1 -1
- package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
- package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
- package/dist/{index-DGIzNtRv.d.ts → index-Dx1kF3Ez.d.ts} +3 -3
- package/dist/{index-DGIzNtRv.d.ts.map → index-Dx1kF3Ez.d.ts.map} +1 -1
- package/dist/index.d.ts +68 -77
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +62 -126
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-DqEsugpr.d.ts} +4 -4
- package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-DqEsugpr.d.ts.map} +1 -1
- package/dist/{integrity-DRXobPEs.d.ts → integrity-CNGUaGBY.d.ts} +3 -3
- package/dist/{integrity-DRXobPEs.d.ts.map → integrity-CNGUaGBY.d.ts.map} +1 -1
- package/dist/{kind-factory-Bvwe3pup.js → kind-factory-BHIgPmzS.js} +2 -2
- package/dist/{kind-factory-Bvwe3pup.js.map → kind-factory-BHIgPmzS.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/{llm-client-D3EoChAU.js → llm-client-Dv5BiKLE.js} +289 -7
- package/dist/llm-client-Dv5BiKLE.js.map +1 -0
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/multishot/index.js +1 -1
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/profile-cell.d.ts +1 -1
- package/dist/{release-report-BZRWdq_t.d.ts → release-report-ChOgpIoQ.d.ts} +4 -4
- package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-ChOgpIoQ.d.ts.map} +1 -1
- package/dist/{replay-CqOsGjzU.js → replay-GW61ezMW.js} +4 -4
- package/dist/{replay-CqOsGjzU.js.map → replay-GW61ezMW.js.map} +1 -1
- package/dist/{replay-DQ-55DC_.d.ts → replay-Krvb114g.d.ts} +7 -7
- package/dist/{replay-DQ-55DC_.d.ts.map → replay-Krvb114g.d.ts.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-BSiCoM1s.d.ts → researcher-xLeNcpKX.d.ts} +6 -6
- package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-xLeNcpKX.d.ts.map} +1 -1
- package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-RZgnGWlx.d.ts} +2 -2
- package/dist/{reward-hacking-DFgkEY4p.d.ts.map → reward-hacking-RZgnGWlx.d.ts.map} +1 -1
- package/dist/rl.d.ts +5 -5
- package/dist/rl.js +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
- package/dist/{run-evidence-H1vRpIdT.d.ts → run-evidence-C6G41MSI.d.ts} +3 -3
- package/dist/{run-evidence-H1vRpIdT.d.ts.map → run-evidence-C6G41MSI.d.ts.map} +1 -1
- package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
- package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-Do5aM9wP.js → semantic-concept-judge-DwF6n05O.js} +3 -3
- package/dist/{semantic-concept-judge-Do5aM9wP.js.map → semantic-concept-judge-DwF6n05O.js.map} +1 -1
- package/dist/{server-Df00sdwz.js → server-D6XJQHw7.js} +2 -2
- package/dist/{server-Df00sdwz.js.map → server-D6XJQHw7.js.map} +1 -1
- package/dist/{skill-usage-DtpLou9L.d.ts → skill-usage-GlOphAhX.d.ts} +9 -9
- package/dist/{skill-usage-DtpLou9L.d.ts.map → skill-usage-GlOphAhX.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-CvSJGdm3.d.ts → skillopt-optimization-method-7S43rbDB.d.ts} +12 -12
- package/dist/{skillopt-optimization-method-CvSJGdm3.d.ts.map → skillopt-optimization-method-7S43rbDB.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-Bfb-vBKe.js} +2 -2
- package/dist/{skillopt-optimization-method-C4FX42dy.js.map → skillopt-optimization-method-Bfb-vBKe.js.map} +1 -1
- package/dist/{statistics-B4u_CiFd.d.ts → statistics-C-dm-J6H.d.ts} +2 -2
- package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-C-dm-J6H.d.ts.map} +1 -1
- package/dist/{store-otlp-D4I90_vR.js → store-otlp-CKtTpRhv.js} +2 -2
- package/dist/{store-otlp-D4I90_vR.js.map → store-otlp-CKtTpRhv.js.map} +1 -1
- package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-B0cAyA7N.d.ts} +3 -3
- package/dist/{summary-report-BOM6dfP7.d.ts.map → summary-report-B0cAyA7N.d.ts.map} +1 -1
- package/dist/{tool-groups-CMmsgTzj.d.ts → tool-groups-CK0JCkqO.d.ts} +9 -9
- package/dist/{tool-groups-CMmsgTzj.d.ts.map → tool-groups-CK0JCkqO.d.ts.map} +1 -1
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +4 -4
- package/dist/{types-y8jrxXWd.d.ts → types-BhP9q0Fq.d.ts} +22 -4
- package/dist/{types-y8jrxXWd.d.ts.map → types-BhP9q0Fq.d.ts.map} +1 -1
- package/dist/{types-DcJxgsLy.d.ts → types-DOZyvsFU.d.ts} +5 -5
- package/dist/{types-DcJxgsLy.d.ts.map → types-DOZyvsFU.d.ts.map} +1 -1
- package/dist/{types-BjMFz88h.d.ts → types-XMVEdrE_.d.ts} +207 -6
- package/dist/types-XMVEdrE_.d.ts.map +1 -0
- package/dist/{usage-receipt-CgxMEBZq.js → usage-receipt-EVI8B8Xu.js} +8 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/building-doctrine.md +15 -0
- package/docs/prime-analyst.md +120 -0
- package/docs/trace-analysis.md +1 -1
- package/package.json +3 -3
- package/dist/benchmark-command-9FTgq6Fg.js.map +0 -1
- package/dist/llm-client-D3EoChAU.js.map +0 -1
- package/dist/types-BjMFz88h.d.ts.map +0 -1
- package/dist/usage-receipt-CgxMEBZq.js.map +0 -1
|
@@ -1,25 +1,27 @@
|
|
|
1
1
|
import { c as ValidationError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
|
|
2
2
|
import { s as resolveModelPricing } from "./metrics-C9YY1OcL.js";
|
|
3
3
|
import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-DMFxsLKr.js";
|
|
4
|
-
import { c as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-
|
|
5
|
-
import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-
|
|
6
|
-
import { o as makeFinding, r as usageReceiptFromCostLedger } from "./usage-receipt-
|
|
4
|
+
import { c as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-Dv5BiKLE.js";
|
|
5
|
+
import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-BHIgPmzS.js";
|
|
6
|
+
import { o as makeFinding, r as usageReceiptFromCostLedger } from "./usage-receipt-EVI8B8Xu.js";
|
|
7
7
|
import { i as hashCanonical, r as canonicalString } from "./canonical-D011XM8r.js";
|
|
8
8
|
import { M as resolveExternalOptimizerProcessLimits, f as runWithCleanup, n as createRunCostLedger, p as startExternalOptimizerModelProxy, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-D5iN0Xzb.js";
|
|
9
|
-
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-
|
|
9
|
+
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-19FQEMBK.js";
|
|
10
10
|
import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-DXZIqu17.js";
|
|
11
11
|
import { E as pairedBootstrap } from "./statistics-ByxzSiOM.js";
|
|
12
|
-
import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-
|
|
13
|
-
import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-
|
|
12
|
+
import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-CKtTpRhv.js";
|
|
13
|
+
import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-B181aMF9.js";
|
|
14
14
|
import { z } from "zod";
|
|
15
15
|
import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
|
|
16
16
|
import * as nodePath from "node:path";
|
|
17
17
|
import { dirname, isAbsolute, relative, resolve, sep } from "node:path";
|
|
18
18
|
import { createHash, randomUUID } from "node:crypto";
|
|
19
|
+
import { request } from "node:http";
|
|
19
20
|
import { link, lstat, mkdir, open, readFile, readdir, realpath, stat, unlink } from "node:fs/promises";
|
|
20
21
|
import { arch, platform } from "node:os";
|
|
21
22
|
import { TextDecoder as TextDecoder$1 } from "node:util";
|
|
22
23
|
import { pathToFileURL } from "node:url";
|
|
24
|
+
import { request as request$1 } from "node:https";
|
|
23
25
|
//#region src/analyst/benchmark-dataset-utils.ts
|
|
24
26
|
function normalizeBenchmarkLabel(value) {
|
|
25
27
|
const normalized = nonEmpty$1(value, "benchmark label").toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g, "");
|
|
@@ -706,7 +708,12 @@ const usageSchema = z.strictObject({
|
|
|
706
708
|
calls: nonNegativeInteger.nullable(),
|
|
707
709
|
tokens: tokenUsageSchema.nullable(),
|
|
708
710
|
cost: costSchema,
|
|
709
|
-
knownCostUsd: nonNegativeNumber.optional()
|
|
711
|
+
knownCostUsd: nonNegativeNumber.optional(),
|
|
712
|
+
partialTokens: z.strictObject({
|
|
713
|
+
input: nonNegativeInteger.nullable(),
|
|
714
|
+
output: nonNegativeInteger.nullable()
|
|
715
|
+
}).optional(),
|
|
716
|
+
tokensEstimated: z.boolean().optional()
|
|
710
717
|
});
|
|
711
718
|
const findingScoreSchema = z.strictObject({
|
|
712
719
|
expectedIssueCount: nonNegativeInteger,
|
|
@@ -1200,7 +1207,7 @@ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
|
|
|
1200
1207
|
"package.json",
|
|
1201
1208
|
"pnpm-lock.yaml"
|
|
1202
1209
|
]);
|
|
1203
|
-
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
1210
|
+
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "c0525dfe7f6931aeefe0e4c98d366cf6558c176e56477c5271d35274f383cf61";
|
|
1204
1211
|
/** The published benchmark evidence was produced at this package version, by
|
|
1205
1212
|
* the retired one-shot direct runner, before trace analysts moved to the
|
|
1206
1213
|
* recursive DSPy RLM engine. Both evidence digests below are historical facts
|
|
@@ -1239,6 +1246,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1239
1246
|
"src/analyst/benchmark-real-model.ts",
|
|
1240
1247
|
"src/analyst/benchmark-report.ts",
|
|
1241
1248
|
"src/analyst/benchmark-response-cache.ts",
|
|
1249
|
+
"src/analyst/benchmark-runner-prime.ts",
|
|
1242
1250
|
"src/analyst/benchmark-scoring.ts",
|
|
1243
1251
|
"src/analyst/benchmark-summary.ts",
|
|
1244
1252
|
"src/analyst/benchmark-verification-artifacts.ts",
|
|
@@ -1251,6 +1259,8 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1251
1259
|
"src/analyst/finding-subject.ts",
|
|
1252
1260
|
"src/analyst/kind-factory.ts",
|
|
1253
1261
|
"src/analyst/parse-tolerant.ts",
|
|
1262
|
+
"src/analyst/prime-bridge-transport.ts",
|
|
1263
|
+
"src/analyst/prime-protocol.ts",
|
|
1254
1264
|
"src/analyst/tool-groups.ts",
|
|
1255
1265
|
"src/analyst/trace-tool-callback.ts",
|
|
1256
1266
|
"src/analyst/types.ts",
|
|
@@ -1269,7 +1279,9 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1269
1279
|
"src/concurrency.ts",
|
|
1270
1280
|
"src/cost-ledger.ts",
|
|
1271
1281
|
"src/errors.ts",
|
|
1282
|
+
"src/integrity/served-model.ts",
|
|
1272
1283
|
"src/judge-calibration.ts",
|
|
1284
|
+
"src/judge-families.ts",
|
|
1273
1285
|
"src/ledger-core/atomic-file-lock.ts",
|
|
1274
1286
|
"src/ledger-core/canonical.ts",
|
|
1275
1287
|
"src/ledger-core/deep-freeze.ts",
|
|
@@ -1299,7 +1311,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1299
1311
|
"src/trace/raw-provider-sink.ts",
|
|
1300
1312
|
"src/verdict-cache.ts"
|
|
1301
1313
|
]);
|
|
1302
|
-
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
1314
|
+
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "47592eaceab13f9cf40530e02a4af3d9ff5bcd6a7eb7d92b110508cc6cd323e5";
|
|
1303
1315
|
function analystBenchmarkImplementationDigest() {
|
|
1304
1316
|
return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256;
|
|
1305
1317
|
}
|
|
@@ -2437,7 +2449,7 @@ function createLocalRunReceipt(config, paths) {
|
|
|
2437
2449
|
traceDir: resolve(config.traceDir),
|
|
2438
2450
|
...config.artifactDir ? { artifactDir: resolve(config.artifactDir) } : {},
|
|
2439
2451
|
outputDir: paths.directory,
|
|
2440
|
-
modelOwnerModule: config.modelOwnerModule
|
|
2452
|
+
...config.modelOwnerModule === void 0 ? {} : { modelOwnerModule: config.modelOwnerModule }
|
|
2441
2453
|
},
|
|
2442
2454
|
command: config.command,
|
|
2443
2455
|
environment: {
|
|
@@ -3596,7 +3608,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
3596
3608
|
const maxModelRequestBytes = config.maxModelRequestBytes ?? 16 * 1024 * 1024;
|
|
3597
3609
|
const maxModelResponseBytes = config.maxModelResponseBytes ?? 4 * 1024 * 1024;
|
|
3598
3610
|
const modelRequestTimeoutMs = config.modelRequestTimeoutMs ?? timeoutMs;
|
|
3599
|
-
const pricing = config.pricing ?? pricingForModel$
|
|
3611
|
+
const pricing = config.pricing ?? pricingForModel$2(model);
|
|
3600
3612
|
const maxCostUsd = config.maxCostUsdPerAnalysis ?? 1;
|
|
3601
3613
|
const costLedger = config.costLedger ?? new CostLedger();
|
|
3602
3614
|
const durability = config.durability ? {
|
|
@@ -3608,7 +3620,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
3608
3620
|
return {
|
|
3609
3621
|
id: "direct",
|
|
3610
3622
|
async analyze(input, context) {
|
|
3611
|
-
const trajectoryId = trajectoryIdFromCaseId$
|
|
3623
|
+
const trajectoryId = trajectoryIdFromCaseId$2(dataset, context.caseId);
|
|
3612
3624
|
const costTags = {
|
|
3613
3625
|
analystId: actor,
|
|
3614
3626
|
benchmarkCaseId: context.caseId,
|
|
@@ -3920,7 +3932,7 @@ function costReceiptMetadata(receipt) {
|
|
|
3920
3932
|
estimatedCostUsd: null
|
|
3921
3933
|
};
|
|
3922
3934
|
}
|
|
3923
|
-
function pricingForModel$
|
|
3935
|
+
function pricingForModel$2(model) {
|
|
3924
3936
|
const pricing = resolveModelPricing(model);
|
|
3925
3937
|
if (!pricing) throw new Error(`no pricing is configured for '${model}'; provide PublicAnalystBenchmarkModelConfig.pricing`);
|
|
3926
3938
|
return {
|
|
@@ -4096,7 +4108,7 @@ async function prepareSingleTraceContext(store, context) {
|
|
|
4096
4108
|
});
|
|
4097
4109
|
}
|
|
4098
4110
|
}
|
|
4099
|
-
function trajectoryIdFromCaseId$
|
|
4111
|
+
function trajectoryIdFromCaseId$2(dataset, caseId) {
|
|
4100
4112
|
const prefix = dataset === "agentrx" ? "agentrx:" : "codetrace:";
|
|
4101
4113
|
if (!caseId.startsWith(prefix) || caseId.length === prefix.length) throw new Error(`unexpected ${dataset} benchmark case id '${caseId}'`);
|
|
4102
4114
|
return caseId.slice(prefix.length);
|
|
@@ -4245,7 +4257,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4245
4257
|
maxToolCalls: config.dspyRlm?.maxToolCalls ?? 80,
|
|
4246
4258
|
maxOutputChars: config.dspyRlm?.maxOutputChars ?? 8e3
|
|
4247
4259
|
};
|
|
4248
|
-
const pricing = config.pricing ?? pricingForModel(config.model);
|
|
4260
|
+
const pricing = config.pricing ?? pricingForModel$1(config.model);
|
|
4249
4261
|
const engine = createDspyRlmTraceEngine({
|
|
4250
4262
|
call: config.call,
|
|
4251
4263
|
callRef: config.callRef,
|
|
@@ -4278,7 +4290,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4278
4290
|
return {
|
|
4279
4291
|
id: "dspy-rlm",
|
|
4280
4292
|
async analyze(input, context) {
|
|
4281
|
-
const trajectoryId = trajectoryIdFromCaseId(dataset, context.caseId);
|
|
4293
|
+
const trajectoryId = trajectoryIdFromCaseId$1(dataset, context.caseId);
|
|
4282
4294
|
const tags = {
|
|
4283
4295
|
benchmarkCaseId: context.caseId,
|
|
4284
4296
|
benchmarkRepetition: String(context.repetition)
|
|
@@ -4518,7 +4530,7 @@ function publicBenchmarkDefinition(dataset, limits, instructions) {
|
|
|
4518
4530
|
limits
|
|
4519
4531
|
};
|
|
4520
4532
|
}
|
|
4521
|
-
function pricingForModel(model) {
|
|
4533
|
+
function pricingForModel$1(model) {
|
|
4522
4534
|
const pricing = resolveModelPricing(model);
|
|
4523
4535
|
if (!pricing) throw new Error(`no pricing is configured for '${model}'; provide PublicAnalystBenchmarkModelConfig.pricing`);
|
|
4524
4536
|
return {
|
|
@@ -4526,7 +4538,7 @@ function pricingForModel(model) {
|
|
|
4526
4538
|
outputUsdPerMillion: pricing.output * 1e3
|
|
4527
4539
|
};
|
|
4528
4540
|
}
|
|
4529
|
-
function trajectoryIdFromCaseId(dataset, caseId) {
|
|
4541
|
+
function trajectoryIdFromCaseId$1(dataset, caseId) {
|
|
4530
4542
|
const prefix = dataset === "agentrx" ? "agentrx:" : "codetrace:";
|
|
4531
4543
|
if (!caseId.startsWith(prefix) || caseId.length === prefix.length) throw new Error(`unexpected ${dataset} benchmark case id '${caseId}'`);
|
|
4532
4544
|
return caseId.slice(prefix.length);
|
|
@@ -4843,6 +4855,832 @@ function rootAgent(row) {
|
|
|
4843
4855
|
return scalarDistributionValue(row.failures.find((failure) => String(failure.failure_id) === String(rootCauseId))?.failed_agent);
|
|
4844
4856
|
}
|
|
4845
4857
|
//#endregion
|
|
4858
|
+
//#region src/analyst/prime-bridge-transport.ts
|
|
4859
|
+
/**
|
|
4860
|
+
* Default transport on node:http/node:https rather than fetch: undici's fixed
|
|
4861
|
+
* response-header timeout kills prime calls that legitimately run past five
|
|
4862
|
+
* minutes, so the request's AbortSignal is the only deadline.
|
|
4863
|
+
*/
|
|
4864
|
+
function nodeHttpPrimeBridgeTransport() {
|
|
4865
|
+
return ({ url, body, signal }) => {
|
|
4866
|
+
const target = new URL(url);
|
|
4867
|
+
if (target.protocol !== "http:" && target.protocol !== "https:") throw new TypeError(`bridge URL must be http: or https:, got ${target.protocol}`);
|
|
4868
|
+
const send = target.protocol === "https:" ? request$1 : request;
|
|
4869
|
+
const encoded = JSON.stringify(body);
|
|
4870
|
+
return new Promise((resolvePromise, rejectPromise) => {
|
|
4871
|
+
const req = send({
|
|
4872
|
+
hostname: target.hostname,
|
|
4873
|
+
port: target.port,
|
|
4874
|
+
path: `${target.pathname}${target.search}`,
|
|
4875
|
+
method: "POST",
|
|
4876
|
+
headers: {
|
|
4877
|
+
"content-type": "application/json",
|
|
4878
|
+
"content-length": Buffer.byteLength(encoded)
|
|
4879
|
+
},
|
|
4880
|
+
signal
|
|
4881
|
+
}, (res) => {
|
|
4882
|
+
const chunks = [];
|
|
4883
|
+
res.on("data", (chunk) => chunks.push(chunk));
|
|
4884
|
+
res.on("end", () => resolvePromise({
|
|
4885
|
+
status: res.statusCode ?? 0,
|
|
4886
|
+
text: Buffer.concat(chunks).toString("utf8")
|
|
4887
|
+
}));
|
|
4888
|
+
res.on("error", rejectPromise);
|
|
4889
|
+
});
|
|
4890
|
+
req.on("error", rejectPromise);
|
|
4891
|
+
req.end(encoded);
|
|
4892
|
+
});
|
|
4893
|
+
};
|
|
4894
|
+
}
|
|
4895
|
+
//#endregion
|
|
4896
|
+
//#region src/analyst/prime-protocol.ts
|
|
4897
|
+
function buildPrimePrompt(spec) {
|
|
4898
|
+
return [
|
|
4899
|
+
`QUESTION: ${spec.question}`,
|
|
4900
|
+
"",
|
|
4901
|
+
...spec.taskDefinition === void 0 ? [] : [
|
|
4902
|
+
"TASK DEFINITION:",
|
|
4903
|
+
spec.taskDefinition,
|
|
4904
|
+
""
|
|
4905
|
+
],
|
|
4906
|
+
...spec.contractLines,
|
|
4907
|
+
"",
|
|
4908
|
+
spec.trajectoryHeader,
|
|
4909
|
+
spec.renderedTrajectory,
|
|
4910
|
+
...spec.trailer === void 0 ? [] : ["", spec.trailer]
|
|
4911
|
+
].join("\n");
|
|
4912
|
+
}
|
|
4913
|
+
/** Carries the malformed reply and the contract — never the trajectory. */
|
|
4914
|
+
function buildPrimeRepairPrompt(spec) {
|
|
4915
|
+
return [
|
|
4916
|
+
"Your previous reply to a trace-analysis task was structurally malformed and could not be parsed",
|
|
4917
|
+
`(${spec.defect}). Below is your previous reply verbatim. Re-emit ONLY the corrected JSON — one`,
|
|
4918
|
+
"fenced ```json block, no other text, no tools. The JSON object has exactly two fields:",
|
|
4919
|
+
...spec.repairContractLines,
|
|
4920
|
+
"",
|
|
4921
|
+
"PREVIOUS REPLY:",
|
|
4922
|
+
spec.previousReply
|
|
4923
|
+
].join("\n");
|
|
4924
|
+
}
|
|
4925
|
+
/**
|
|
4926
|
+
* Recover the reply's JSON object.
|
|
4927
|
+
*
|
|
4928
|
+
* Distinct from `extractJsonPayload` in ../llm-client, which serves a response
|
|
4929
|
+
* that DECLARES a JSON root and therefore must not scan onward. A prime reply
|
|
4930
|
+
* is prose plus a fenced block, and when the model emits several fences the
|
|
4931
|
+
* last one is its answer — so fences are scanned in reverse, and only then is a
|
|
4932
|
+
* brace-to-brace slice tried.
|
|
4933
|
+
*/
|
|
4934
|
+
function extractPrimeJsonObject(text) {
|
|
4935
|
+
const direct = parsePrimeJsonObject(text);
|
|
4936
|
+
if (direct) return direct;
|
|
4937
|
+
const fenced = [...text.matchAll(/```(?:json)?\s*\n?([\s\S]*?)```/g)];
|
|
4938
|
+
for (let index = fenced.length - 1; index >= 0; index -= 1) {
|
|
4939
|
+
const candidate = parsePrimeJsonObject(fenced[index][1]);
|
|
4940
|
+
if (candidate) return candidate;
|
|
4941
|
+
}
|
|
4942
|
+
const start = text.indexOf("{");
|
|
4943
|
+
const end = text.lastIndexOf("}");
|
|
4944
|
+
if (start >= 0 && end > start) {
|
|
4945
|
+
const candidate = parsePrimeJsonObject(text.slice(start, end + 1));
|
|
4946
|
+
if (candidate) return candidate;
|
|
4947
|
+
}
|
|
4948
|
+
return null;
|
|
4949
|
+
}
|
|
4950
|
+
/** Why the reply cannot be read as a prime answer, or null when it can. */
|
|
4951
|
+
function primeReplyDefect(parsed, rowsField) {
|
|
4952
|
+
if (parsed === null) return "no parseable JSON object";
|
|
4953
|
+
if (!Array.isArray(parsed[rowsField])) return `JSON has no "${rowsField}" array`;
|
|
4954
|
+
return null;
|
|
4955
|
+
}
|
|
4956
|
+
function parsePrimeJsonObject(text) {
|
|
4957
|
+
try {
|
|
4958
|
+
const value = JSON.parse(text.trim());
|
|
4959
|
+
return typeof value === "object" && value !== null && !Array.isArray(value) ? value : null;
|
|
4960
|
+
} catch {
|
|
4961
|
+
return null;
|
|
4962
|
+
}
|
|
4963
|
+
}
|
|
4964
|
+
function emptyPrimeRawUsage() {
|
|
4965
|
+
return {
|
|
4966
|
+
calls: null,
|
|
4967
|
+
inputTokens: null,
|
|
4968
|
+
outputTokens: null,
|
|
4969
|
+
bridgeEstimated: false
|
|
4970
|
+
};
|
|
4971
|
+
}
|
|
4972
|
+
/** Read the bridge's OpenAI-shaped `usage` object. */
|
|
4973
|
+
function normalizePrimeUsage(raw) {
|
|
4974
|
+
if (typeof raw !== "object" || raw === null) return emptyPrimeRawUsage();
|
|
4975
|
+
const record = raw;
|
|
4976
|
+
return {
|
|
4977
|
+
calls: tokenCountOrNull(record.model_requests),
|
|
4978
|
+
inputTokens: tokenCountOrNull(record.prompt_tokens),
|
|
4979
|
+
outputTokens: tokenCountOrNull(record.completion_tokens),
|
|
4980
|
+
bridgeEstimated: record.estimated === true
|
|
4981
|
+
};
|
|
4982
|
+
}
|
|
4983
|
+
/**
|
|
4984
|
+
* Sum two turns. Each side poisons independently: two turns that both report
|
|
4985
|
+
* input and neither report output yield a real input total beside a null
|
|
4986
|
+
* output, because discarding a measured count is as wrong as inventing one.
|
|
4987
|
+
*/
|
|
4988
|
+
function mergePrimeRawUsage(a, b) {
|
|
4989
|
+
return {
|
|
4990
|
+
calls: sumOrNull(a.calls, b.calls),
|
|
4991
|
+
inputTokens: sumOrNull(a.inputTokens, b.inputTokens),
|
|
4992
|
+
outputTokens: sumOrNull(a.outputTokens, b.outputTokens),
|
|
4993
|
+
bridgeEstimated: a.bridgeEstimated || b.bridgeEstimated
|
|
4994
|
+
};
|
|
4995
|
+
}
|
|
4996
|
+
function sumOrNull(a, b) {
|
|
4997
|
+
return a !== null && b !== null ? a + b : null;
|
|
4998
|
+
}
|
|
4999
|
+
function tokenCountOrNull(value) {
|
|
5000
|
+
return typeof value === "number" && Number.isSafeInteger(value) && value >= 0 ? value : null;
|
|
5001
|
+
}
|
|
5002
|
+
/**
|
|
5003
|
+
* Bind raw prime usage to agent-eval's typed receipt.
|
|
5004
|
+
*
|
|
5005
|
+
* `RunTokenUsage.input` and `.output` are non-nullable, so a one-sided count
|
|
5006
|
+
* cannot round-trip through `tokens` without writing a zero nobody measured.
|
|
5007
|
+
* The complete-accounting field therefore stays null, the reported side is
|
|
5008
|
+
* carried verbatim in `partialTokens`, and its price becomes the receipt's
|
|
5009
|
+
* `knownCostUsd` lower bound.
|
|
5010
|
+
*
|
|
5011
|
+
* Only agent-eval calls this; consumers with no pricing table read
|
|
5012
|
+
* `PrimeRawUsage` directly.
|
|
5013
|
+
*/
|
|
5014
|
+
function analystUsageReceiptFromPrimeUsage(usage, pricing) {
|
|
5015
|
+
const { calls, inputTokens, outputTokens, bridgeEstimated } = usage;
|
|
5016
|
+
const estimatedTokens = bridgeEstimated ? { tokensEstimated: true } : {};
|
|
5017
|
+
if (inputTokens !== null && outputTokens !== null) return {
|
|
5018
|
+
calls,
|
|
5019
|
+
tokens: {
|
|
5020
|
+
input: inputTokens,
|
|
5021
|
+
output: outputTokens
|
|
5022
|
+
},
|
|
5023
|
+
cost: {
|
|
5024
|
+
kind: "estimated",
|
|
5025
|
+
usd: priceTokens(inputTokens, outputTokens, pricing)
|
|
5026
|
+
},
|
|
5027
|
+
...estimatedTokens
|
|
5028
|
+
};
|
|
5029
|
+
if (inputTokens === null && outputTokens === null) return {
|
|
5030
|
+
calls,
|
|
5031
|
+
tokens: null,
|
|
5032
|
+
cost: {
|
|
5033
|
+
kind: "uncaptured",
|
|
5034
|
+
usd: null
|
|
5035
|
+
},
|
|
5036
|
+
...estimatedTokens
|
|
5037
|
+
};
|
|
5038
|
+
return {
|
|
5039
|
+
calls,
|
|
5040
|
+
tokens: null,
|
|
5041
|
+
partialTokens: {
|
|
5042
|
+
input: inputTokens,
|
|
5043
|
+
output: outputTokens
|
|
5044
|
+
},
|
|
5045
|
+
cost: {
|
|
5046
|
+
kind: "uncaptured",
|
|
5047
|
+
usd: null
|
|
5048
|
+
},
|
|
5049
|
+
knownCostUsd: priceTokens(inputTokens ?? 0, outputTokens ?? 0, pricing),
|
|
5050
|
+
...estimatedTokens
|
|
5051
|
+
};
|
|
5052
|
+
}
|
|
5053
|
+
function priceTokens(input, output, pricing) {
|
|
5054
|
+
return (input * pricing.inputUsdPerMillion + output * pricing.outputUsdPerMillion) / 1e6;
|
|
5055
|
+
}
|
|
5056
|
+
/**
|
|
5057
|
+
* Run the protocol: one call, one bounded repair turn on a structurally
|
|
5058
|
+
* malformed reply, then decode. Zero valid rows from a well-formed reply is an
|
|
5059
|
+
* honest null, not a failure.
|
|
5060
|
+
*/
|
|
5061
|
+
async function runPrimeExchange(options) {
|
|
5062
|
+
const { contract } = options;
|
|
5063
|
+
const turns = [];
|
|
5064
|
+
const repair = {
|
|
5065
|
+
attempted: false,
|
|
5066
|
+
succeeded: null
|
|
5067
|
+
};
|
|
5068
|
+
const first = await callPrimeTurn(options, options.prompt);
|
|
5069
|
+
if (!first.ok) return {
|
|
5070
|
+
ok: false,
|
|
5071
|
+
failure: first.failure,
|
|
5072
|
+
usage: mergeTurns(turns),
|
|
5073
|
+
turns,
|
|
5074
|
+
repair
|
|
5075
|
+
};
|
|
5076
|
+
turns.push({
|
|
5077
|
+
turn: "first",
|
|
5078
|
+
usage: first.usage,
|
|
5079
|
+
rawUsage: first.rawUsage
|
|
5080
|
+
});
|
|
5081
|
+
let reply = first.content;
|
|
5082
|
+
let parsed = extractPrimeJsonObject(reply);
|
|
5083
|
+
let defect = primeReplyDefect(parsed, contract.rowsField);
|
|
5084
|
+
if (defect !== null && options.repair) {
|
|
5085
|
+
repair.attempted = true;
|
|
5086
|
+
const second = await callPrimeTurn(options, buildPrimeRepairPrompt({
|
|
5087
|
+
defect,
|
|
5088
|
+
previousReply: reply,
|
|
5089
|
+
repairContractLines: contract.repairContractLines
|
|
5090
|
+
}));
|
|
5091
|
+
if (!second.ok) return {
|
|
5092
|
+
ok: false,
|
|
5093
|
+
failure: second.failure,
|
|
5094
|
+
usage: mergeTurns(turns),
|
|
5095
|
+
turns,
|
|
5096
|
+
repair,
|
|
5097
|
+
reply
|
|
5098
|
+
};
|
|
5099
|
+
turns.push({
|
|
5100
|
+
turn: "repair",
|
|
5101
|
+
usage: second.usage,
|
|
5102
|
+
rawUsage: second.rawUsage
|
|
5103
|
+
});
|
|
5104
|
+
reply = second.content;
|
|
5105
|
+
parsed = extractPrimeJsonObject(reply);
|
|
5106
|
+
defect = primeReplyDefect(parsed, contract.rowsField);
|
|
5107
|
+
repair.succeeded = defect === null;
|
|
5108
|
+
}
|
|
5109
|
+
const usage = mergeTurns(turns);
|
|
5110
|
+
if (defect !== null) return {
|
|
5111
|
+
ok: false,
|
|
5112
|
+
failure: {
|
|
5113
|
+
kind: "malformed-reply",
|
|
5114
|
+
message: `${defect} in prime reply${repair.attempted ? " even after the bounded repair turn" : ""}`
|
|
5115
|
+
},
|
|
5116
|
+
usage,
|
|
5117
|
+
turns,
|
|
5118
|
+
repair,
|
|
5119
|
+
reply
|
|
5120
|
+
};
|
|
5121
|
+
const rawRows = parsed[contract.rowsField];
|
|
5122
|
+
const rows = [];
|
|
5123
|
+
const rejected = [];
|
|
5124
|
+
let overflow = 0;
|
|
5125
|
+
rawRows.forEach((row, index) => {
|
|
5126
|
+
const decoded = contract.decodeRow(row, index);
|
|
5127
|
+
if (!decoded.ok) {
|
|
5128
|
+
rejected.push({
|
|
5129
|
+
index,
|
|
5130
|
+
reason: decoded.reason
|
|
5131
|
+
});
|
|
5132
|
+
return;
|
|
5133
|
+
}
|
|
5134
|
+
if (contract.maxRows !== void 0 && rows.length >= contract.maxRows) {
|
|
5135
|
+
overflow += 1;
|
|
5136
|
+
return;
|
|
5137
|
+
}
|
|
5138
|
+
rows.push(decoded.row);
|
|
5139
|
+
});
|
|
5140
|
+
const answer = parsed.answer;
|
|
5141
|
+
return {
|
|
5142
|
+
ok: true,
|
|
5143
|
+
answer: typeof answer === "string" ? answer : null,
|
|
5144
|
+
rows,
|
|
5145
|
+
rejected,
|
|
5146
|
+
reportedRows: rawRows.length,
|
|
5147
|
+
overflow,
|
|
5148
|
+
usage,
|
|
5149
|
+
turns,
|
|
5150
|
+
repair,
|
|
5151
|
+
reply
|
|
5152
|
+
};
|
|
5153
|
+
}
|
|
5154
|
+
async function callPrimeTurn(options, content) {
|
|
5155
|
+
const { transport, url, model, timeoutMs, signal } = options;
|
|
5156
|
+
const controller = new AbortController();
|
|
5157
|
+
const forwardAbort = () => controller.abort(signal?.reason);
|
|
5158
|
+
if (signal?.aborted) controller.abort(signal.reason);
|
|
5159
|
+
else signal?.addEventListener("abort", forwardAbort, { once: true });
|
|
5160
|
+
const deadline = setTimeout(() => controller.abort(), timeoutMs);
|
|
5161
|
+
let result;
|
|
5162
|
+
try {
|
|
5163
|
+
result = await transport({
|
|
5164
|
+
url,
|
|
5165
|
+
body: {
|
|
5166
|
+
model,
|
|
5167
|
+
messages: [{
|
|
5168
|
+
role: "user",
|
|
5169
|
+
content
|
|
5170
|
+
}]
|
|
5171
|
+
},
|
|
5172
|
+
signal: controller.signal
|
|
5173
|
+
});
|
|
5174
|
+
} catch (error) {
|
|
5175
|
+
if (signal?.aborted) return {
|
|
5176
|
+
ok: false,
|
|
5177
|
+
failure: {
|
|
5178
|
+
kind: "aborted",
|
|
5179
|
+
message: "prime exchange cancelled by the caller",
|
|
5180
|
+
cause: error
|
|
5181
|
+
}
|
|
5182
|
+
};
|
|
5183
|
+
if (controller.signal.aborted) return {
|
|
5184
|
+
ok: false,
|
|
5185
|
+
failure: {
|
|
5186
|
+
kind: "deadline",
|
|
5187
|
+
message: `bridge call exceeded ${timeoutMs}ms`
|
|
5188
|
+
}
|
|
5189
|
+
};
|
|
5190
|
+
return {
|
|
5191
|
+
ok: false,
|
|
5192
|
+
failure: {
|
|
5193
|
+
kind: "transport",
|
|
5194
|
+
message: `bridge transport failure: ${error instanceof Error ? error.message : String(error)}`
|
|
5195
|
+
}
|
|
5196
|
+
};
|
|
5197
|
+
} finally {
|
|
5198
|
+
clearTimeout(deadline);
|
|
5199
|
+
signal?.removeEventListener("abort", forwardAbort);
|
|
5200
|
+
}
|
|
5201
|
+
if (result.status !== 200) {
|
|
5202
|
+
const bodySnippet = result.text.slice(0, 500);
|
|
5203
|
+
return {
|
|
5204
|
+
ok: false,
|
|
5205
|
+
failure: {
|
|
5206
|
+
kind: "http-status",
|
|
5207
|
+
message: `bridge HTTP ${result.status}: ${bodySnippet}`,
|
|
5208
|
+
status: result.status,
|
|
5209
|
+
bodySnippet
|
|
5210
|
+
}
|
|
5211
|
+
};
|
|
5212
|
+
}
|
|
5213
|
+
let response;
|
|
5214
|
+
try {
|
|
5215
|
+
response = JSON.parse(result.text);
|
|
5216
|
+
} catch {
|
|
5217
|
+
return {
|
|
5218
|
+
ok: false,
|
|
5219
|
+
failure: {
|
|
5220
|
+
kind: "unparseable-json",
|
|
5221
|
+
message: `bridge returned unparseable JSON (${result.text.length} bytes)`
|
|
5222
|
+
}
|
|
5223
|
+
};
|
|
5224
|
+
}
|
|
5225
|
+
const replyContent = primeReplyContent(response);
|
|
5226
|
+
if (replyContent === null) return {
|
|
5227
|
+
ok: false,
|
|
5228
|
+
failure: {
|
|
5229
|
+
kind: "no-content",
|
|
5230
|
+
message: "bridge reply carries no message content"
|
|
5231
|
+
}
|
|
5232
|
+
};
|
|
5233
|
+
const rawUsage = primeReplyUsage(response);
|
|
5234
|
+
return {
|
|
5235
|
+
ok: true,
|
|
5236
|
+
content: replyContent,
|
|
5237
|
+
usage: normalizePrimeUsage(rawUsage),
|
|
5238
|
+
rawUsage
|
|
5239
|
+
};
|
|
5240
|
+
}
|
|
5241
|
+
/**
|
|
5242
|
+
* Fold from the FIRST turn, never from an empty receipt: an all-null identity
|
|
5243
|
+
* would poison every side it merged with and erase counts the bridge reported.
|
|
5244
|
+
*/
|
|
5245
|
+
function mergeTurns(turns) {
|
|
5246
|
+
if (turns.length === 0) return emptyPrimeRawUsage();
|
|
5247
|
+
return turns.slice(1).reduce((total, turn) => mergePrimeRawUsage(total, turn.usage), turns[0].usage);
|
|
5248
|
+
}
|
|
5249
|
+
function primeReplyContent(response) {
|
|
5250
|
+
if (typeof response !== "object" || response === null) return null;
|
|
5251
|
+
const choices = response.choices;
|
|
5252
|
+
if (!Array.isArray(choices) || choices.length === 0) return null;
|
|
5253
|
+
const message = choices[0]?.message;
|
|
5254
|
+
if (typeof message !== "object" || message === null) return null;
|
|
5255
|
+
const content = message.content;
|
|
5256
|
+
return typeof content === "string" && content.length > 0 ? content : null;
|
|
5257
|
+
}
|
|
5258
|
+
function primeReplyUsage(response) {
|
|
5259
|
+
if (typeof response !== "object" || response === null) return null;
|
|
5260
|
+
return response.usage ?? null;
|
|
5261
|
+
}
|
|
5262
|
+
/**
|
|
5263
|
+
* Render, measure, fall back to the capped projection, re-measure, fail loud.
|
|
5264
|
+
*
|
|
5265
|
+
* Inline is the only delivery prime has, so an oversized trajectory is a
|
|
5266
|
+
* refusal rather than a silent truncation: dropping spans would understate the
|
|
5267
|
+
* trajectory and the analyst would answer a question about a different run.
|
|
5268
|
+
*/
|
|
5269
|
+
async function projectPrimeTrajectory(source, limits) {
|
|
5270
|
+
let fetch = "full";
|
|
5271
|
+
let items = await source.full();
|
|
5272
|
+
if (items === null) {
|
|
5273
|
+
fetch = "capped";
|
|
5274
|
+
items = await source.capped();
|
|
5275
|
+
}
|
|
5276
|
+
let rendered = JSON.stringify(items);
|
|
5277
|
+
if (rendered.length > limits.maxInlineChars && fetch === "full") {
|
|
5278
|
+
fetch = "capped";
|
|
5279
|
+
items = await source.capped();
|
|
5280
|
+
rendered = JSON.stringify(items);
|
|
5281
|
+
}
|
|
5282
|
+
if (rendered.length > limits.maxInlineChars) return {
|
|
5283
|
+
ok: false,
|
|
5284
|
+
reason: `trajectory renders to ${rendered.length} chars even at ${source.cappedDescription}; inline delivery impossible`,
|
|
5285
|
+
renderedChars: rendered.length
|
|
5286
|
+
};
|
|
5287
|
+
return {
|
|
5288
|
+
ok: true,
|
|
5289
|
+
items,
|
|
5290
|
+
rendered,
|
|
5291
|
+
delivery: {
|
|
5292
|
+
mode: "inline-json",
|
|
5293
|
+
fetch,
|
|
5294
|
+
renderedChars: rendered.length
|
|
5295
|
+
}
|
|
5296
|
+
};
|
|
5297
|
+
}
|
|
5298
|
+
/**
|
|
5299
|
+
* Digest of everything a consumer can send to the bridge under the prime
|
|
5300
|
+
* protocol, recorded per observation so a prime result names the exact contract
|
|
5301
|
+
* that produced it.
|
|
5302
|
+
*
|
|
5303
|
+
* Computed over the ACTUALLY composed contract, so two consumers that both
|
|
5304
|
+
* stamp `analyst_id: 'prime'` while asking materially different questions get
|
|
5305
|
+
* different digests by construction. That is what makes 'prime' a reproducible
|
|
5306
|
+
* claim rather than a label.
|
|
5307
|
+
*/
|
|
5308
|
+
function primeProtocolSha256(identity) {
|
|
5309
|
+
return createHash("sha256").update(JSON.stringify({
|
|
5310
|
+
kind: "prime-analyst-protocol",
|
|
5311
|
+
question: identity.question,
|
|
5312
|
+
taskPrompt: identity.taskDefinition ?? null,
|
|
5313
|
+
outputContract: identity.contractLines,
|
|
5314
|
+
repairContract: buildPrimeRepairPrompt({
|
|
5315
|
+
defect: "<defect>",
|
|
5316
|
+
previousReply: "<previous-reply>",
|
|
5317
|
+
repairContractLines: identity.repairContractLines
|
|
5318
|
+
}),
|
|
5319
|
+
limits: identity.limits
|
|
5320
|
+
})).digest("hex");
|
|
5321
|
+
}
|
|
5322
|
+
//#endregion
|
|
5323
|
+
//#region src/analyst/benchmark-runner-prime.ts
|
|
5324
|
+
/**
|
|
5325
|
+
* Prime analyst arm: the RLM coding agent reached through an OpenAI-compatible
|
|
5326
|
+
* cli-bridge solves the CodeTraceBench incorrect-step task as a one-shot trace
|
|
5327
|
+
* analyst.
|
|
5328
|
+
*
|
|
5329
|
+
* The runner consumes the same prepared benchmark cases every other runner
|
|
5330
|
+
* receives — the trace store already carries the appended final-verification
|
|
5331
|
+
* spans — and produces findings through the same published block expansion, so
|
|
5332
|
+
* a prime observation and a dspy-rlm observation differ only in which analyst
|
|
5333
|
+
* produced the blocks.
|
|
5334
|
+
*
|
|
5335
|
+
* The protocol itself — prompt composition, the bounded repair turn, reply
|
|
5336
|
+
* extraction, the projection ladder, usage normalization — lives in
|
|
5337
|
+
* `./prime-protocol`, which knows nothing about CodeTraceBench. This file is
|
|
5338
|
+
* the benchmark's binding to it: the block row grammar, the store-backed
|
|
5339
|
+
* projection source, and the benchmark observation shape.
|
|
5340
|
+
*
|
|
5341
|
+
* Trajectory delivery is inline JSON in the prompt. The dspy typed path binds
|
|
5342
|
+
* the viewTrace span projection as a REPL variable; prime has no REPL, so the
|
|
5343
|
+
* same projection is serialized into the prompt. When the full projection is
|
|
5344
|
+
* oversized the runner falls back to chunked viewSpans over the same
|
|
5345
|
+
* projection surface with a per-attribute byte cap, and fails loud if the
|
|
5346
|
+
* result still exceeds the inline budget.
|
|
5347
|
+
*
|
|
5348
|
+
* A structurally malformed reply gets ONE bounded repair turn (disable with
|
|
5349
|
+
* `repair: false`): a second stateless call carrying the malformed reply plus
|
|
5350
|
+
* the output contract — never the trajectory — mirroring the dspy arm's typed
|
|
5351
|
+
* repair so both arms face the same structured-output affordance. Still
|
|
5352
|
+
* malformed after repair = failed observation with a typed error, exactly how
|
|
5353
|
+
* a dspy-rlm failure is recorded. Zero valid blocks from a well-formed reply
|
|
5354
|
+
* is an honest null, not a failure.
|
|
5355
|
+
*/
|
|
5356
|
+
const PRIME_ANALYST_ID = "prime";
|
|
5357
|
+
const PRIME_QUESTION = "Which assistant steps are incorrect under the CodeTraceBench definition?";
|
|
5358
|
+
/** Ceiling on the serialized trajectory JSON embedded in the prompt. */
|
|
5359
|
+
const MAX_INLINE_TRAJECTORY_CHARS = 36e4;
|
|
5360
|
+
/** Per-attribute projection cap used by the chunked viewSpans fallback. */
|
|
5361
|
+
const CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP = 1200;
|
|
5362
|
+
/** Minimal per-attribute cap used only to enumerate span ids in store order. */
|
|
5363
|
+
const SPAN_ID_ENUMERATION_ATTRIBUTE_BYTE_CAP = 64;
|
|
5364
|
+
/** viewSpans accepts at most 100 ids per call; 40 keeps each response bounded. */
|
|
5365
|
+
const VIEW_SPANS_CHUNK_SIZE = 40;
|
|
5366
|
+
const PRIME_SEVERITIES = /* @__PURE__ */ new Set([
|
|
5367
|
+
"critical",
|
|
5368
|
+
"high",
|
|
5369
|
+
"medium",
|
|
5370
|
+
"low",
|
|
5371
|
+
"info"
|
|
5372
|
+
]);
|
|
5373
|
+
var PrimeBridgeTransportError = class extends Error {};
|
|
5374
|
+
var PrimeBridgeHttpError = class extends Error {
|
|
5375
|
+
status;
|
|
5376
|
+
constructor(status, bodySnippet) {
|
|
5377
|
+
super(`bridge HTTP ${status}: ${bodySnippet}`);
|
|
5378
|
+
this.status = status;
|
|
5379
|
+
}
|
|
5380
|
+
};
|
|
5381
|
+
var PrimeMalformedReplyError = class extends Error {};
|
|
5382
|
+
var PrimeTraceProjectionError = class extends Error {};
|
|
5383
|
+
/**
|
|
5384
|
+
* Short-strings rule: long reply strings get corrupted when the bridge splices
|
|
5385
|
+
* its backend's stream, so the contract forbids a rationale field and caps
|
|
5386
|
+
* every string the model must emit.
|
|
5387
|
+
*/
|
|
5388
|
+
const PRIME_OUTPUT_CONTRACT_LINES = [
|
|
5389
|
+
"OUTPUT CONTRACT (supersedes any transport wording above — you have no trace tools and no REPL):",
|
|
5390
|
+
"You are a one-shot analyst. Every fact you need is in the TRAJECTORY JSON below.",
|
|
5391
|
+
"Do not run shell commands, do not read or write files, do not use any tools.",
|
|
5392
|
+
"Reply with EXACTLY one fenced ```json code block and no other fenced block. The JSON object has exactly two fields:",
|
|
5393
|
+
" \"answer\": string — ONE short sentence (max 300 chars) naming the latest failure evidence you traced from.",
|
|
5394
|
+
" \"blocks\": array (possibly empty) of failure blocks, each exactly:",
|
|
5395
|
+
" {\"first_step\": int, \"last_step\": int, \"consequence_step\": int,",
|
|
5396
|
+
" \"escape_status\": \"escaped\"|\"unescaped\",",
|
|
5397
|
+
" \"severity\": \"critical\"|\"high\"|\"medium\"|\"low\"|\"info\",",
|
|
5398
|
+
" \"claim\": string (ONE short sentence, max 200 chars),",
|
|
5399
|
+
" \"confidence\": number 0..1}",
|
|
5400
|
+
"Do NOT include a rationale field. Keep every string SHORT — long strings get corrupted in transport and void your work.",
|
|
5401
|
+
`Report at most 16 blocks; a block spans at most 12 steps.`,
|
|
5402
|
+
"Every step number must be the n of an existing assistant span with span_id \"step-<n>\" and kind \"LLM\" in the trajectory below; never cite TOOL, CHAIN, or AGENT spans.",
|
|
5403
|
+
"\"blocks\" is [] only for a clean trajectory."
|
|
5404
|
+
];
|
|
5405
|
+
const PRIME_REPAIR_CONTRACT_LINES = [
|
|
5406
|
+
" \"answer\": string (ONE short sentence, max 300 chars)",
|
|
5407
|
+
" \"blocks\": array (possibly empty) of {\"first_step\": int, \"last_step\": int, \"consequence_step\": int,",
|
|
5408
|
+
" \"escape_status\": \"escaped\"|\"unescaped\", \"severity\": \"critical\"|\"high\"|\"medium\"|\"low\"|\"info\",",
|
|
5409
|
+
" \"claim\": string (max 200 chars), \"confidence\": number 0..1}",
|
|
5410
|
+
"No rationale field. Keep every string SHORT. Preserve the step numbers and verdicts of your previous reply exactly; shorten prose freely."
|
|
5411
|
+
];
|
|
5412
|
+
const PRIME_PROTOCOL_IDENTITY = {
|
|
5413
|
+
question: PRIME_QUESTION,
|
|
5414
|
+
taskDefinition: CODE_TRACE_BENCH_ANALYST_PROMPT,
|
|
5415
|
+
contractLines: PRIME_OUTPUT_CONTRACT_LINES,
|
|
5416
|
+
repairContractLines: PRIME_REPAIR_CONTRACT_LINES,
|
|
5417
|
+
limits: {
|
|
5418
|
+
maxBlocks: 16,
|
|
5419
|
+
maxBlockSteps: 12,
|
|
5420
|
+
maxInlineTrajectoryChars: MAX_INLINE_TRAJECTORY_CHARS,
|
|
5421
|
+
chunkedProjectionAttributeByteCap: CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP
|
|
5422
|
+
}
|
|
5423
|
+
};
|
|
5424
|
+
/**
|
|
5425
|
+
* The block row grammar. No `maxRows`: the count cap belongs to
|
|
5426
|
+
* `expandCodeTraceFailureBlocks`, which drops the offending block and names it
|
|
5427
|
+
* in `diagnostics.droppedBlocks`, so capping here would erase that record.
|
|
5428
|
+
*/
|
|
5429
|
+
const PRIME_BLOCK_CONTRACT = {
|
|
5430
|
+
rowsField: "blocks",
|
|
5431
|
+
contractLines: PRIME_OUTPUT_CONTRACT_LINES,
|
|
5432
|
+
repairContractLines: PRIME_REPAIR_CONTRACT_LINES,
|
|
5433
|
+
decodeRow(row) {
|
|
5434
|
+
const reason = blockRowDefect(row);
|
|
5435
|
+
if (reason !== null) return {
|
|
5436
|
+
ok: false,
|
|
5437
|
+
reason
|
|
5438
|
+
};
|
|
5439
|
+
return {
|
|
5440
|
+
ok: true,
|
|
5441
|
+
row: blockFromRow(row)
|
|
5442
|
+
};
|
|
5443
|
+
}
|
|
5444
|
+
};
|
|
5445
|
+
/**
|
|
5446
|
+
* Digest of everything this runner can send to the bridge, recorded per
|
|
5447
|
+
* observation so a prime result names the exact contract that produced it.
|
|
5448
|
+
*/
|
|
5449
|
+
function primeAnalystProtocolSha256() {
|
|
5450
|
+
return primeProtocolSha256(PRIME_PROTOCOL_IDENTITY);
|
|
5451
|
+
}
|
|
5452
|
+
/** CodeTraceBench-only: the prompt and output contract speak its block grammar. */
|
|
5453
|
+
function createPrimeBenchmarkRunner(options) {
|
|
5454
|
+
const baseUrl = requiredString(options.baseUrl, "baseUrl").replace(/\/+$/, "");
|
|
5455
|
+
const model = requiredString(options.model, "model");
|
|
5456
|
+
const timeoutMs = positiveSafeInteger(options.timeoutMs, "timeoutMs");
|
|
5457
|
+
const repair = options.repair;
|
|
5458
|
+
if (typeof repair !== "boolean") throw new TypeError("repair must be a boolean");
|
|
5459
|
+
const pricing = options.pricing ?? pricingForModel(model);
|
|
5460
|
+
const transport = options.transport ?? nodeHttpPrimeBridgeTransport();
|
|
5461
|
+
const url = `${baseUrl}/v1/chat/completions`;
|
|
5462
|
+
return {
|
|
5463
|
+
id: PRIME_ANALYST_ID,
|
|
5464
|
+
async analyze(input, context) {
|
|
5465
|
+
const trajectoryId = trajectoryIdFromCaseId(context.caseId);
|
|
5466
|
+
let usage;
|
|
5467
|
+
let metadata = {
|
|
5468
|
+
analysisMode: "prime-rlm",
|
|
5469
|
+
engine: "prime",
|
|
5470
|
+
bridgeUrl: baseUrl,
|
|
5471
|
+
model,
|
|
5472
|
+
protocolSha256: primeAnalystProtocolSha256()
|
|
5473
|
+
};
|
|
5474
|
+
try {
|
|
5475
|
+
const store = input.traceStore;
|
|
5476
|
+
if (!store) throw new Error("codetracebench prime runner requires a trace store");
|
|
5477
|
+
const projection = await projectPrimeTrajectory(codeTraceProjectionSource(store, trajectoryId, context.signal ? { signal: context.signal } : void 0), { maxInlineChars: MAX_INLINE_TRAJECTORY_CHARS });
|
|
5478
|
+
if (!projection.ok) throw new PrimeTraceProjectionError(projection.reason);
|
|
5479
|
+
const delivery = {
|
|
5480
|
+
mode: projection.delivery.mode,
|
|
5481
|
+
fetch: projection.delivery.fetch === "full" ? "view-trace" : "view-spans-chunked",
|
|
5482
|
+
perAttributeByteCap: projection.delivery.fetch === "full" ? null : CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP,
|
|
5483
|
+
renderedChars: projection.delivery.renderedChars
|
|
5484
|
+
};
|
|
5485
|
+
metadata = {
|
|
5486
|
+
...metadata,
|
|
5487
|
+
delivery
|
|
5488
|
+
};
|
|
5489
|
+
const prompt = buildCodeTracePrompt(trajectoryId, projection.items, projection.rendered);
|
|
5490
|
+
metadata = {
|
|
5491
|
+
...metadata,
|
|
5492
|
+
promptChars: prompt.length
|
|
5493
|
+
};
|
|
5494
|
+
const outcome = await runPrimeExchange({
|
|
5495
|
+
contract: PRIME_BLOCK_CONTRACT,
|
|
5496
|
+
prompt,
|
|
5497
|
+
transport,
|
|
5498
|
+
url,
|
|
5499
|
+
model,
|
|
5500
|
+
timeoutMs,
|
|
5501
|
+
repair,
|
|
5502
|
+
...context.signal ? { signal: context.signal } : {}
|
|
5503
|
+
});
|
|
5504
|
+
if (!outcome.ok && outcome.failure.kind === "aborted") throw abortCause(outcome.failure);
|
|
5505
|
+
if (outcome.turns.length > 0) {
|
|
5506
|
+
usage = analystUsageReceiptFromPrimeUsage(outcome.usage, pricing);
|
|
5507
|
+
metadata = {
|
|
5508
|
+
...metadata,
|
|
5509
|
+
bridgeUsage: bridgeUsageFromTurns(outcome.turns)
|
|
5510
|
+
};
|
|
5511
|
+
}
|
|
5512
|
+
metadata = {
|
|
5513
|
+
...metadata,
|
|
5514
|
+
repair: outcome.repair
|
|
5515
|
+
};
|
|
5516
|
+
if (!outcome.ok) {
|
|
5517
|
+
if (outcome.reply !== void 0) metadata = {
|
|
5518
|
+
...metadata,
|
|
5519
|
+
reply: outcome.reply.slice(0, 4e3)
|
|
5520
|
+
};
|
|
5521
|
+
throw primeFailureError(outcome.failure);
|
|
5522
|
+
}
|
|
5523
|
+
const expanded = await expandCodeTraceFailureBlocks({
|
|
5524
|
+
trajectoryId,
|
|
5525
|
+
blocks: outcome.rows,
|
|
5526
|
+
store,
|
|
5527
|
+
analystId: PRIME_ANALYST_ID,
|
|
5528
|
+
...context.signal ? { signal: context.signal } : {}
|
|
5529
|
+
});
|
|
5530
|
+
return {
|
|
5531
|
+
findings: expanded.findings,
|
|
5532
|
+
usage,
|
|
5533
|
+
metadata: {
|
|
5534
|
+
...metadata,
|
|
5535
|
+
answer: outcome.answer,
|
|
5536
|
+
reportedRows: outcome.reportedRows,
|
|
5537
|
+
rejectedRows: outcome.rejected,
|
|
5538
|
+
blockDiagnostics: expanded.diagnostics
|
|
5539
|
+
}
|
|
5540
|
+
};
|
|
5541
|
+
} catch (error) {
|
|
5542
|
+
if (context.signal?.aborted) throw error;
|
|
5543
|
+
return {
|
|
5544
|
+
findings: [],
|
|
5545
|
+
...usage ? { usage } : {},
|
|
5546
|
+
error: publicBenchmarkError(error, []),
|
|
5547
|
+
metadata
|
|
5548
|
+
};
|
|
5549
|
+
}
|
|
5550
|
+
}
|
|
5551
|
+
};
|
|
5552
|
+
}
|
|
5553
|
+
/**
|
|
5554
|
+
* The trace store, seen through the protocol's two-move projection contract:
|
|
5555
|
+
* the full viewTrace projection, or the chunked viewSpans projection at a
|
|
5556
|
+
* per-attribute byte cap.
|
|
5557
|
+
*/
|
|
5558
|
+
function codeTraceProjectionSource(store, trajectoryId, context) {
|
|
5559
|
+
return {
|
|
5560
|
+
async full() {
|
|
5561
|
+
return (await store.viewTrace({ trace_id: trajectoryId }, context)).spans ?? null;
|
|
5562
|
+
},
|
|
5563
|
+
capped: () => projectSpansChunked(store, trajectoryId, context),
|
|
5564
|
+
cappedDescription: `per-attribute cap ${CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP}`
|
|
5565
|
+
};
|
|
5566
|
+
}
|
|
5567
|
+
/**
|
|
5568
|
+
* Chunked viewSpans projection for traces whose full viewTrace response is
|
|
5569
|
+
* oversized. Span ids come from a minimal-cap viewTrace in store order; every
|
|
5570
|
+
* id must project or the case fails loud — a silently dropped span would
|
|
5571
|
+
* understate the trajectory.
|
|
5572
|
+
*/
|
|
5573
|
+
async function projectSpansChunked(store, trajectoryId, context) {
|
|
5574
|
+
const enumeration = await store.viewTrace({
|
|
5575
|
+
trace_id: trajectoryId,
|
|
5576
|
+
per_attribute_byte_cap: SPAN_ID_ENUMERATION_ATTRIBUTE_BYTE_CAP
|
|
5577
|
+
}, context);
|
|
5578
|
+
if (!enumeration.spans) throw new PrimeTraceProjectionError(`trace '${trajectoryId}' is oversized even at per-attribute cap ${SPAN_ID_ENUMERATION_ATTRIBUTE_BYTE_CAP}; cannot enumerate span ids`);
|
|
5579
|
+
const ids = [];
|
|
5580
|
+
const seen = /* @__PURE__ */ new Set();
|
|
5581
|
+
for (const span of enumeration.spans) if (typeof span.span_id === "string" && span.span_id.length > 0 && !seen.has(span.span_id)) {
|
|
5582
|
+
seen.add(span.span_id);
|
|
5583
|
+
ids.push(span.span_id);
|
|
5584
|
+
}
|
|
5585
|
+
if (ids.length === 0) throw new PrimeTraceProjectionError(`no span ids parsed from trace '${trajectoryId}'`);
|
|
5586
|
+
const projected = [];
|
|
5587
|
+
for (let index = 0; index < ids.length; index += VIEW_SPANS_CHUNK_SIZE) {
|
|
5588
|
+
const chunk = ids.slice(index, index + VIEW_SPANS_CHUNK_SIZE);
|
|
5589
|
+
const result = await store.viewSpans({
|
|
5590
|
+
trace_id: trajectoryId,
|
|
5591
|
+
span_ids: chunk,
|
|
5592
|
+
per_attribute_byte_cap: CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP
|
|
5593
|
+
}, context);
|
|
5594
|
+
if (result.missing_span_ids.length > 0 || result.omitted_span_ids.length > 0 || result.spans.length !== chunk.length) throw new PrimeTraceProjectionError(`viewSpans projected ${result.spans.length}/${chunk.length} spans for chunk at ${index} of '${trajectoryId}'`);
|
|
5595
|
+
projected.push(...result.spans);
|
|
5596
|
+
}
|
|
5597
|
+
return projected;
|
|
5598
|
+
}
|
|
5599
|
+
function buildCodeTracePrompt(trajectoryId, spans, renderedSpans) {
|
|
5600
|
+
const stepSpans = spans.filter((span) => /^step-\d+$/.test(String(span.span_id)));
|
|
5601
|
+
if (stepSpans.length === 0) throw new PrimeTraceProjectionError(`no step-<n> spans in trace '${trajectoryId}'`);
|
|
5602
|
+
const finalVerification = spans.filter(isFinalVerificationSpan);
|
|
5603
|
+
return buildPrimePrompt({
|
|
5604
|
+
question: PRIME_QUESTION,
|
|
5605
|
+
taskDefinition: CODE_TRACE_BENCH_ANALYST_PROMPT,
|
|
5606
|
+
contractLines: PRIME_OUTPUT_CONTRACT_LINES,
|
|
5607
|
+
trajectoryHeader: `TRAJECTORY (trace_id ${trajectoryId}; ${stepSpans.length} assistant step spans; full span projection as JSON):`,
|
|
5608
|
+
renderedTrajectory: renderedSpans,
|
|
5609
|
+
trailer: finalVerification.length > 0 ? `FINAL VERIFICATION SPANS:\n${JSON.stringify(finalVerification)}` : "FINAL VERIFICATION: unavailable for this trajectory — trace backward from the latest failure evidence inside the trajectory itself."
|
|
5610
|
+
});
|
|
5611
|
+
}
|
|
5612
|
+
/** Map the protocol's terminal reason onto this benchmark's typed error classes. */
|
|
5613
|
+
function primeFailureError(failure) {
|
|
5614
|
+
switch (failure.kind) {
|
|
5615
|
+
case "http-status": return new PrimeBridgeHttpError(failure.status, failure.bodySnippet);
|
|
5616
|
+
case "malformed-reply": return new PrimeMalformedReplyError(failure.message);
|
|
5617
|
+
default: return new PrimeBridgeTransportError(failure.message);
|
|
5618
|
+
}
|
|
5619
|
+
}
|
|
5620
|
+
/** A cancelled run is not a result: the caller's error propagates unchanged. */
|
|
5621
|
+
function abortCause(failure) {
|
|
5622
|
+
return failure.cause instanceof Error ? failure.cause : new Error(failure.message);
|
|
5623
|
+
}
|
|
5624
|
+
function bridgeUsageFromTurns(turns) {
|
|
5625
|
+
return {
|
|
5626
|
+
first: turns.find((turn) => turn.turn === "first")?.rawUsage ?? null,
|
|
5627
|
+
repair: turns.find((turn) => turn.turn === "repair")?.rawUsage ?? null
|
|
5628
|
+
};
|
|
5629
|
+
}
|
|
5630
|
+
function isFinalVerificationSpan(span) {
|
|
5631
|
+
if (span.span_id.startsWith("benchmark-verification")) return true;
|
|
5632
|
+
const role = span.attributes["benchmark.evidence.role"];
|
|
5633
|
+
return typeof role === "string" && role.startsWith("final-verification");
|
|
5634
|
+
}
|
|
5635
|
+
function trajectoryIdFromCaseId(caseId) {
|
|
5636
|
+
if (!caseId.startsWith("codetrace:") || caseId.length === 10) throw new Error(`unexpected codetracebench benchmark case id '${caseId}'`);
|
|
5637
|
+
return caseId.slice(10);
|
|
5638
|
+
}
|
|
5639
|
+
function blockRowDefect(row) {
|
|
5640
|
+
if (typeof row !== "object" || row === null || Array.isArray(row)) return "row is not an object";
|
|
5641
|
+
const record = row;
|
|
5642
|
+
for (const field of [
|
|
5643
|
+
"first_step",
|
|
5644
|
+
"last_step",
|
|
5645
|
+
"consequence_step"
|
|
5646
|
+
]) {
|
|
5647
|
+
const value = record[field];
|
|
5648
|
+
if (!Number.isInteger(value) || value < 1) return `${field} must be a positive integer`;
|
|
5649
|
+
}
|
|
5650
|
+
const firstStep = record.first_step;
|
|
5651
|
+
const lastStep = record.last_step;
|
|
5652
|
+
const consequenceStep = record.consequence_step;
|
|
5653
|
+
if (lastStep < firstStep) return "last_step < first_step";
|
|
5654
|
+
if (consequenceStep < firstStep) return "consequence_step < first_step";
|
|
5655
|
+
if (lastStep - firstStep + 1 > 12) return `block spans ${lastStep - firstStep + 1} steps (cap 12)`;
|
|
5656
|
+
if (record.escape_status !== "escaped" && record.escape_status !== "unescaped") return "escape_status must be escaped|unescaped";
|
|
5657
|
+
if (typeof record.severity !== "string" || !PRIME_SEVERITIES.has(record.severity)) return "severity outside the analyst severity enum";
|
|
5658
|
+
if (typeof record.claim !== "string" || record.claim.trim().length === 0 || record.claim.length > 2e3) return "claim must be a 1-2000 char string";
|
|
5659
|
+
if (typeof record.confidence !== "number" || !Number.isFinite(record.confidence) || record.confidence < 0 || record.confidence > 1) return "confidence must be 0..1";
|
|
5660
|
+
return null;
|
|
5661
|
+
}
|
|
5662
|
+
function blockFromRow(row) {
|
|
5663
|
+
const rationale = typeof row.rationale === "string" && row.rationale.trim().length > 0 ? row.rationale.trim().slice(0, 4e3) : void 0;
|
|
5664
|
+
return {
|
|
5665
|
+
firstStep: row.first_step,
|
|
5666
|
+
lastStep: row.last_step,
|
|
5667
|
+
consequenceStep: row.consequence_step,
|
|
5668
|
+
escapeStatus: row.escape_status,
|
|
5669
|
+
severity: row.severity,
|
|
5670
|
+
claim: row.claim.trim(),
|
|
5671
|
+
confidence: row.confidence,
|
|
5672
|
+
...rationale === void 0 ? {} : { rationale }
|
|
5673
|
+
};
|
|
5674
|
+
}
|
|
5675
|
+
function pricingForModel(model) {
|
|
5676
|
+
const pricing = resolveModelPricing(model);
|
|
5677
|
+
if (!pricing) throw new Error(`no pricing is configured for '${model}'; provide PrimeBenchmarkRunnerOptions.pricing`);
|
|
5678
|
+
return {
|
|
5679
|
+
inputUsdPerMillion: pricing.input * 1e3,
|
|
5680
|
+
outputUsdPerMillion: pricing.output * 1e3
|
|
5681
|
+
};
|
|
5682
|
+
}
|
|
5683
|
+
//#endregion
|
|
4846
5684
|
//#region src/analyst/benchmark-report.ts
|
|
4847
5685
|
function renderAnalystBenchmarkMarkdown(result, comparisons = []) {
|
|
4848
5686
|
const { provenance } = result;
|
|
@@ -4973,7 +5811,20 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
|
|
|
4973
5811
|
return benchmarkExitCode(artifact.result, config.analyst);
|
|
4974
5812
|
}
|
|
4975
5813
|
if (await regularFileExists(paths.report)) throw new Error(`benchmark report exists without a completed result; refusing ambiguous resume: ${paths.report}`);
|
|
4976
|
-
const createAnalystRunner = dependencies.createAnalystRunner ?? ((dataset, model) =>
|
|
5814
|
+
const createAnalystRunner = dependencies.createAnalystRunner ?? ((dataset, model) => {
|
|
5815
|
+
if (config.analyst === "prime") {
|
|
5816
|
+
if (!config.prime) throw new Error("analyst 'prime' is missing its bridge configuration");
|
|
5817
|
+
return createPrimeBenchmarkRunner({
|
|
5818
|
+
baseUrl: config.prime.bridgeUrl,
|
|
5819
|
+
model: model.model,
|
|
5820
|
+
timeoutMs: model.timeoutMs,
|
|
5821
|
+
repair: config.prime.repair,
|
|
5822
|
+
...model.pricing ? { pricing: model.pricing } : {}
|
|
5823
|
+
});
|
|
5824
|
+
}
|
|
5825
|
+
const ownerModel = requireModelOwnerSettings(model);
|
|
5826
|
+
return config.analyst === "direct" ? createPublicBenchmarkDirectRunner(dataset, ownerModel) : createPublicBenchmarkRlmRunner(dataset, ownerModel);
|
|
5827
|
+
});
|
|
4977
5828
|
const runners = [emptyPublicBenchmarkRunner(), createAnalystRunner(config.dataset, {
|
|
4978
5829
|
...config.model,
|
|
4979
5830
|
costLedger,
|
|
@@ -5143,21 +5994,31 @@ Run the recursive DSPy trace analyst against public AgentRx or CodeTraceBench la
|
|
|
5143
5994
|
|
|
5144
5995
|
Required:
|
|
5145
5996
|
--dataset agentrx|codetracebench
|
|
5146
|
-
--analyst dspy-rlm|direct
|
|
5997
|
+
--analyst dspy-rlm|direct|prime Scored analyst. Default: dspy-rlm.
|
|
5147
5998
|
'direct' is the one-shot comparison arm.
|
|
5999
|
+
'prime' is the RLM coding agent behind an
|
|
6000
|
+
OpenAI-compatible cli-bridge (codetracebench
|
|
6001
|
+
only; see docs/prime-analyst.md)
|
|
5148
6002
|
--labels <dataset.json|dataset.jsonl>
|
|
5149
6003
|
--trace-dir <one-trace-per-file OTLP JSONL directory>
|
|
5150
6004
|
--artifact-dir <extracted artifact root> Required for CodeTraceBench
|
|
5151
6005
|
--out <new output directory>
|
|
5152
6006
|
--revision <full 40- or 64-character hex digest>
|
|
5153
6007
|
--split <dataset split>
|
|
5154
|
-
--model-owner-module <module> Module exporting
|
|
5155
|
-
the owner keeps
|
|
5156
|
-
|
|
6008
|
+
--model-owner-module <module> dspy-rlm|direct only. Module exporting
|
|
6009
|
+
createModelExecutionOwner; the owner keeps
|
|
6010
|
+
provider credentials and policy
|
|
6011
|
+
--model <provider model id> For prime, the bridge model id in
|
|
6012
|
+
<backend>/<provider>/<model> form, e.g.
|
|
6013
|
+
prime/zai/glm-5.2
|
|
5157
6014
|
--limit <positive case count>
|
|
5158
6015
|
|
|
5159
6016
|
Controls:
|
|
5160
6017
|
--resume Continue an interrupted run in --out
|
|
6018
|
+
--bridge-url <url> prime only. OpenAI-compatible cli-bridge
|
|
6019
|
+
base URL. Default: http://localhost:4181
|
|
6020
|
+
--no-repair prime only. Disable the bounded repair turn
|
|
6021
|
+
for a structurally malformed reply
|
|
5161
6022
|
--seed <integer> Case-selection and comparison seed. Default: 0
|
|
5162
6023
|
--concurrency <positive integer> Parallel benchmark jobs. Default: 1
|
|
5163
6024
|
--repetitions <positive integer> Runs per case and runner. Default: 1
|
|
@@ -5207,7 +6068,17 @@ async function parseCommandConfig(argv, env, dependencies) {
|
|
|
5207
6068
|
const maxCostUsd = positiveFiniteFlag(flags, "max-cost-usd", 5);
|
|
5208
6069
|
const python = flags.get("python")?.trim();
|
|
5209
6070
|
const analyst = flags.get("analyst")?.trim() ?? "dspy-rlm";
|
|
5210
|
-
if (analyst !== "dspy-rlm" && analyst !== "direct") throw new Error("--analyst must be 'dspy-rlm' or '
|
|
6071
|
+
if (analyst !== "dspy-rlm" && analyst !== "direct" && analyst !== "prime") throw new Error("--analyst must be 'dspy-rlm', 'direct', or 'prime'");
|
|
6072
|
+
const bridgeUrl = flags.get("bridge-url")?.trim();
|
|
6073
|
+
if (bridgeUrl !== void 0 && analyst !== "prime") throw new Error("--bridge-url requires --analyst prime");
|
|
6074
|
+
if (bridgeUrl === "") throw new Error("--bridge-url must not be blank");
|
|
6075
|
+
if (flags.has("no-repair") && analyst !== "prime") throw new Error("--no-repair requires --analyst prime");
|
|
6076
|
+
if (analyst === "prime" && dataset !== "codetracebench") throw new Error("--analyst prime requires --dataset codetracebench; the prime runner speaks the CodeTraceBench failure-block contract");
|
|
6077
|
+
if (analyst === "prime" && flags.has("model-owner-module")) throw new Error("--model-owner-module is not used by --analyst prime; the cli-bridge owns model execution");
|
|
6078
|
+
const prime = analyst === "prime" ? {
|
|
6079
|
+
bridgeUrl: bridgeUrl ?? "http://localhost:4181",
|
|
6080
|
+
repair: !flags.has("no-repair")
|
|
6081
|
+
} : void 0;
|
|
5211
6082
|
const rlmSamples = positiveFlag(flags, "rlm-samples", 1);
|
|
5212
6083
|
if (rlmSamples > 1 && analyst !== "dspy-rlm") throw new Error("--rlm-samples above 1 requires --analyst dspy-rlm");
|
|
5213
6084
|
const instructionsFile = flags.get("instructions-file")?.trim();
|
|
@@ -5215,13 +6086,13 @@ async function parseCommandConfig(argv, env, dependencies) {
|
|
|
5215
6086
|
const instructionsOverride = instructionsFile ? readAnalystInstructionsOverride(instructionsFile) : void 0;
|
|
5216
6087
|
if (rlmSamples > 1 && dataset !== "codetracebench") throw new Error("--rlm-samples above 1 requires --dataset codetracebench; step-level consensus is defined on its block grammar");
|
|
5217
6088
|
const model = requiredFlag(flags, "model");
|
|
5218
|
-
const modelOwnerModule = requiredFlag(flags, "model-owner-module");
|
|
5219
|
-
const owner = await (dependencies.loadModelExecutionOwner ?? loadModelExecutionOwner)(modelOwnerModule, {
|
|
6089
|
+
const modelOwnerModule = analyst === "prime" ? void 0 : requiredFlag(flags, "model-owner-module");
|
|
6090
|
+
const owner = modelOwnerModule ? await (dependencies.loadModelExecutionOwner ?? loadModelExecutionOwner)(modelOwnerModule, {
|
|
5220
6091
|
model,
|
|
5221
6092
|
environment: Object.freeze({ ...env })
|
|
5222
|
-
});
|
|
5223
|
-
assertModelExecutionOwner(owner);
|
|
5224
|
-
const pricing = owner
|
|
6093
|
+
}) : void 0;
|
|
6094
|
+
if (owner) assertModelExecutionOwner(owner);
|
|
6095
|
+
const pricing = owner?.pricing ?? benchmarkModelPricing(model);
|
|
5225
6096
|
const maxOutputTokens = positiveFlag(flags, "max-output-tokens", 16384);
|
|
5226
6097
|
const timeoutMs = positiveFlag(flags, "timeout-ms", 3e5);
|
|
5227
6098
|
return {
|
|
@@ -5234,9 +6105,11 @@ async function parseCommandConfig(argv, env, dependencies) {
|
|
|
5234
6105
|
revision: immutableRevision(requiredFlag(flags, "revision")),
|
|
5235
6106
|
split: requiredFlag(flags, "split"),
|
|
5236
6107
|
model: {
|
|
5237
|
-
|
|
5238
|
-
|
|
5239
|
-
|
|
6108
|
+
...owner ? {
|
|
6109
|
+
call: owner.call,
|
|
6110
|
+
callRef: owner.callRef,
|
|
6111
|
+
recordExecution: owner.recordExecution
|
|
6112
|
+
} : { callRef: `cli-bridge:${prime.bridgeUrl}` },
|
|
5240
6113
|
model,
|
|
5241
6114
|
maxOutputTokens,
|
|
5242
6115
|
timeoutMs,
|
|
@@ -5274,11 +6147,22 @@ async function parseCommandConfig(argv, env, dependencies) {
|
|
|
5274
6147
|
rlmSamples,
|
|
5275
6148
|
maxCostUsd,
|
|
5276
6149
|
maxArtifactBytes: positiveFlag(flags, "max-artifact-bytes", DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES),
|
|
5277
|
-
modelOwnerModule,
|
|
6150
|
+
...modelOwnerModule === void 0 ? {} : { modelOwnerModule },
|
|
6151
|
+
...prime === void 0 ? {} : { prime },
|
|
5278
6152
|
command: `agent-eval analyst-benchmark ${argv.filter((argument) => argument !== "--resume").map(shellQuote).join(" ")}`,
|
|
5279
6153
|
resume: flags.has("resume")
|
|
5280
6154
|
};
|
|
5281
6155
|
}
|
|
6156
|
+
/** Fail-loud narrowing: the dspy-rlm and direct analysts require an owner call path. */
|
|
6157
|
+
function requireModelOwnerSettings(model) {
|
|
6158
|
+
const { call, recordExecution } = model;
|
|
6159
|
+
if (typeof call !== "function" || typeof recordExecution !== "function") throw new Error("model-owner execution is required for the dspy-rlm and direct analysts");
|
|
6160
|
+
return {
|
|
6161
|
+
...model,
|
|
6162
|
+
call,
|
|
6163
|
+
recordExecution
|
|
6164
|
+
};
|
|
6165
|
+
}
|
|
5282
6166
|
function parseFlags(argv) {
|
|
5283
6167
|
const flags = /* @__PURE__ */ new Map();
|
|
5284
6168
|
for (let index = 0; index < argv.length; index += 1) {
|
|
@@ -5304,6 +6188,8 @@ const KNOWN_FLAGS = /* @__PURE__ */ new Set([
|
|
|
5304
6188
|
"resume",
|
|
5305
6189
|
"dataset",
|
|
5306
6190
|
"analyst",
|
|
6191
|
+
"bridge-url",
|
|
6192
|
+
"no-repair",
|
|
5307
6193
|
"labels",
|
|
5308
6194
|
"trace-dir",
|
|
5309
6195
|
"artifact-dir",
|
|
@@ -5339,7 +6225,7 @@ const KNOWN_FLAGS = /* @__PURE__ */ new Set([
|
|
|
5339
6225
|
"max-cost-usd",
|
|
5340
6226
|
"max-artifact-bytes"
|
|
5341
6227
|
]);
|
|
5342
|
-
const BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["resume"]);
|
|
6228
|
+
const BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["resume", "no-repair"]);
|
|
5343
6229
|
function assertKnownFlags(flags) {
|
|
5344
6230
|
for (const flag of flags.keys()) if (!KNOWN_FLAGS.has(flag)) throw new Error(`unknown analyst-benchmark flag: --${flag}`);
|
|
5345
6231
|
}
|
|
@@ -5467,6 +6353,6 @@ function shellQuote(value) {
|
|
|
5467
6353
|
return /^[a-zA-Z0-9_./:@%+=,-]+$/.test(value) ? value : `'${value.replaceAll("'", "'\\''")}'`;
|
|
5468
6354
|
}
|
|
5469
6355
|
//#endregion
|
|
5470
|
-
export {
|
|
6356
|
+
export { ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as $, summarizeCodeTraceCalibration as A, publicBenchmarkSystemPrompt as B, createPublicBenchmarkRlmRunner as C, expandCodeTraceFailureBlocks as D, emptyPublicBenchmarkRunner as E, CODE_TRACE_BENCH_ANALYST_PROMPT as F, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as G, appendVerificationArtifactsToOtlp as H, MAX_INCORRECT_BLOCKS as I, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as J, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as K, MAX_INCORRECT_BLOCK_STEPS as L, analystInstructionsOverrideFromText as M, effectiveAnalystProtocolSha256 as N, readAnalystBenchmarkArtifact as O, readAnalystInstructionsOverride as P, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as Q, publicBenchmarkProtocolSha256 as R, selectPublicBenchmarkRows as S, adaptPublicBenchmarkFindings as T, loadCodeTraceVerificationArtifacts as U, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as V, parseVerificationOutcome as W, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as X, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as Y, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as Z, nodeHttpPrimeBridgeTransport as _, primeAnalystProtocolSha256 as a, ANALYST_BENCHMARK_OBSERVATIONS_FILE as at, publicBenchmarkDistributions as b, buildPrimeRepairPrompt as c, summarizeAgentRxCalibration as ct, mergePrimeRawUsage as d, agentRxBenchmarkCase as dt, analystBenchmarkDependencyLockDigest as et, normalizePrimeUsage as f, agentRxPredictionsToFindings as ft, runPrimeExchange as g, projectPrimeTrajectory as h, normalizeBenchmarkLabel as ht, createPrimeBenchmarkRunner as i, ANALYST_BENCHMARK_MANIFEST_FILE as it, compareAnalystRunners as j, renderCodeTraceCalibrationMarkdown as k, emptyPrimeRawUsage as l, codeTraceBenchCase as lt, primeReplyDefect as m, roundAgentRxStep as mt, runAnalystBenchmarkCommand as n, ANALYST_BENCHMARK_COST_LEDGER_FILE as nt, analystUsageReceiptFromPrimeUsage as o, AGENT_RX_UPSTREAM_REVISION as ot, primeProtocolSha256 as p, normalizeAgentRxCategory as pt, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as q, renderAnalystBenchmarkMarkdown as r, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as rt, buildPrimePrompt as s, renderAgentRxCalibrationMarkdown as st, ANALYST_BENCHMARK_HELP as t, analystBenchmarkImplementationDigest as tt, extractPrimeJsonObject as u, codeTracerPredictionsToFindings as ut, loadPublicBenchmarkRows as v, createPublicBenchmarkDirectRunner as w, publicBenchmarkSelectionReport as x, preparePublicAnalystBenchmark as y, publicBenchmarkRlmInstructions as z };
|
|
5471
6357
|
|
|
5472
|
-
//# sourceMappingURL=benchmark-command-
|
|
6358
|
+
//# sourceMappingURL=benchmark-command-CQd78YHt.js.map
|