@tangle-network/agent-eval 0.144.3 → 0.144.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/dist/analyst/index.d.ts +363 -16
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +8 -8
- package/dist/{benchmark-CYtcIF2V.js → benchmark-B181aMF9.js} +2 -2
- package/dist/{benchmark-CYtcIF2V.js.map → benchmark-B181aMF9.js.map} +1 -1
- package/dist/{benchmark-CP6kWfj8.d.ts → benchmark-Fmo42QVE.d.ts} +3 -3
- package/dist/{benchmark-CP6kWfj8.d.ts.map → benchmark-Fmo42QVE.d.ts.map} +1 -1
- package/dist/{benchmark-command-CQPKRUr-.js → benchmark-command-CA_NFOmy.js} +919 -35
- package/dist/benchmark-command-CA_NFOmy.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CkG1bWFa.js → benchmarks-CWbsj0t4.js} +2 -2
- package/dist/{benchmarks-CkG1bWFa.js.map → benchmarks-CWbsj0t4.js.map} +1 -1
- package/dist/campaign/index.d.ts +4 -4
- package/dist/campaign/index.js +2 -2
- package/dist/{campaign-DjGFyPxH.js → campaign-ClpnD7Ug.js} +2 -2
- package/dist/{campaign-DjGFyPxH.js.map → campaign-ClpnD7Ug.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-Bbht4xxl.d.ts → client-DAb7MWtL.d.ts} +2 -2
- package/dist/{client-Bbht4xxl.d.ts.map → client-DAb7MWtL.d.ts.map} +1 -1
- package/dist/{completion-verifier-EJERfFwF.d.ts → completion-verifier-VvpHRu78.d.ts} +3 -3
- package/dist/{completion-verifier-EJERfFwF.d.ts.map → completion-verifier-VvpHRu78.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +6 -6
- package/dist/contract/index.js +4 -4
- package/dist/control.d.ts +2 -2
- package/dist/{default-registry-iXfu2trt.d.ts → default-registry-BwbZ9N9v.d.ts} +5 -5
- package/dist/{default-registry-iXfu2trt.d.ts.map → default-registry-BwbZ9N9v.d.ts.map} +1 -1
- package/dist/{default-registry-SOyHB6qG.js → default-registry-RLNNoeEP.js} +3 -3
- package/dist/{default-registry-SOyHB6qG.js.map → default-registry-RLNNoeEP.js.map} +1 -1
- package/dist/{dspy-rlm-engine-IRCG8kdi.js → dspy-rlm-engine-19FQEMBK.js} +2 -2
- package/dist/{dspy-rlm-engine-IRCG8kdi.js.map → dspy-rlm-engine-19FQEMBK.js.map} +1 -1
- package/dist/{exact-types-BygCBR4L.d.ts → exact-types-BQ7W90C4.d.ts} +2 -2
- package/dist/{exact-types-BygCBR4L.d.ts.map → exact-types-BQ7W90C4.d.ts.map} +1 -1
- package/dist/{extract-usage-7l1Xq5ti.js → extract-usage-BW27f3XW.js} +2 -2
- package/dist/{extract-usage-7l1Xq5ti.js.map → extract-usage-BW27f3XW.js.map} +1 -1
- package/dist/{feedback-trajectory-CSIkRLQX.d.ts → feedback-trajectory-WK7x4mhy.d.ts} +3 -3
- package/dist/{feedback-trajectory-CSIkRLQX.d.ts.map → feedback-trajectory-WK7x4mhy.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/{index-BuHs_OnD.d.ts → index-CpxZSlB7.d.ts} +6 -6
- package/dist/{index-BuHs_OnD.d.ts.map → index-CpxZSlB7.d.ts.map} +1 -1
- package/dist/{index-D_qTihaQ.d.ts → index-CsuAo2-J.d.ts} +4 -4
- package/dist/{index-D_qTihaQ.d.ts.map → index-CsuAo2-J.d.ts.map} +1 -1
- package/dist/index.d.ts +13 -13
- package/dist/index.js +10 -10
- package/dist/{kind-factory-Bvwe3pup.js → kind-factory-BHIgPmzS.js} +2 -2
- package/dist/{kind-factory-Bvwe3pup.js.map → kind-factory-BHIgPmzS.js.map} +1 -1
- package/dist/multishot/index.d.ts +1 -1
- package/dist/multishot/index.js +1 -1
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{replay-DQ-55DC_.d.ts → replay-B7S7Pdbw.d.ts} +3 -3
- package/dist/{replay-DQ-55DC_.d.ts.map → replay-B7S7Pdbw.d.ts.map} +1 -1
- package/dist/{replay-CqOsGjzU.js → replay-GW61ezMW.js} +4 -4
- package/dist/{replay-CqOsGjzU.js.map → replay-GW61ezMW.js.map} +1 -1
- package/dist/rl.d.ts +1 -1
- package/dist/{run-evidence-H1vRpIdT.d.ts → run-evidence-j5Ynww6L.d.ts} +2 -2
- package/dist/{run-evidence-H1vRpIdT.d.ts.map → run-evidence-j5Ynww6L.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-Do5aM9wP.js → semantic-concept-judge-DKRtp2sY.js} +2 -2
- package/dist/{semantic-concept-judge-Do5aM9wP.js.map → semantic-concept-judge-DKRtp2sY.js.map} +1 -1
- package/dist/{skill-usage-DtpLou9L.d.ts → skill-usage-DUvvudWR.d.ts} +5 -5
- package/dist/{skill-usage-DtpLou9L.d.ts.map → skill-usage-DUvvudWR.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-CvSJGdm3.d.ts → skillopt-optimization-method-CGz9ywhM.d.ts} +4 -4
- package/dist/{skillopt-optimization-method-CvSJGdm3.d.ts.map → skillopt-optimization-method-CGz9ywhM.d.ts.map} +1 -1
- package/dist/{store-otlp-D4I90_vR.js → store-otlp-CKtTpRhv.js} +2 -2
- package/dist/{store-otlp-D4I90_vR.js.map → store-otlp-CKtTpRhv.js.map} +1 -1
- package/dist/{tool-groups-CMmsgTzj.d.ts → tool-groups-ByZiqpVk.d.ts} +4 -4
- package/dist/{tool-groups-CMmsgTzj.d.ts.map → tool-groups-ByZiqpVk.d.ts.map} +1 -1
- package/dist/traces.d.ts +3 -3
- package/dist/traces.js +4 -4
- package/dist/{types-y8jrxXWd.d.ts → types-CZt1PBIk.d.ts} +19 -1
- package/dist/{types-y8jrxXWd.d.ts.map → types-CZt1PBIk.d.ts.map} +1 -1
- package/dist/{types-DcJxgsLy.d.ts → types-Dcoaqcsc.d.ts} +2 -2
- package/dist/{types-DcJxgsLy.d.ts.map → types-Dcoaqcsc.d.ts.map} +1 -1
- package/dist/{usage-receipt-CgxMEBZq.js → usage-receipt-EVI8B8Xu.js} +8 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -0
- package/dist/wire/index.d.ts +1 -1
- package/docs/prime-analyst.md +120 -0
- package/docs/trace-analysis.md +1 -1
- package/package.json +3 -3
- package/dist/benchmark-command-CQPKRUr-.js.map +0 -1
- package/dist/usage-receipt-CgxMEBZq.js.map +0 -1
|
@@ -2,24 +2,26 @@ import { c as ValidationError, t as AgentEvalError } from "./errors-D-LKuDhb.js"
|
|
|
2
2
|
import { s as resolveModelPricing } from "./metrics-C9YY1OcL.js";
|
|
3
3
|
import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-DMFxsLKr.js";
|
|
4
4
|
import { c as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-D3EoChAU.js";
|
|
5
|
-
import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-
|
|
6
|
-
import { o as makeFinding, r as usageReceiptFromCostLedger } from "./usage-receipt-
|
|
5
|
+
import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-BHIgPmzS.js";
|
|
6
|
+
import { o as makeFinding, r as usageReceiptFromCostLedger } from "./usage-receipt-EVI8B8Xu.js";
|
|
7
7
|
import { i as hashCanonical, r as canonicalString } from "./canonical-D011XM8r.js";
|
|
8
8
|
import { M as resolveExternalOptimizerProcessLimits, f as runWithCleanup, n as createRunCostLedger, p as startExternalOptimizerModelProxy, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-D5iN0Xzb.js";
|
|
9
|
-
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-
|
|
9
|
+
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-19FQEMBK.js";
|
|
10
10
|
import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-DXZIqu17.js";
|
|
11
11
|
import { E as pairedBootstrap } from "./statistics-ByxzSiOM.js";
|
|
12
|
-
import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-
|
|
13
|
-
import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-
|
|
12
|
+
import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-CKtTpRhv.js";
|
|
13
|
+
import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-B181aMF9.js";
|
|
14
14
|
import { z } from "zod";
|
|
15
15
|
import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
|
|
16
16
|
import * as nodePath from "node:path";
|
|
17
17
|
import { dirname, isAbsolute, relative, resolve, sep } from "node:path";
|
|
18
18
|
import { createHash, randomUUID } from "node:crypto";
|
|
19
|
+
import { request } from "node:http";
|
|
19
20
|
import { link, lstat, mkdir, open, readFile, readdir, realpath, stat, unlink } from "node:fs/promises";
|
|
20
21
|
import { arch, platform } from "node:os";
|
|
21
22
|
import { TextDecoder as TextDecoder$1 } from "node:util";
|
|
22
23
|
import { pathToFileURL } from "node:url";
|
|
24
|
+
import { request as request$1 } from "node:https";
|
|
23
25
|
//#region src/analyst/benchmark-dataset-utils.ts
|
|
24
26
|
function normalizeBenchmarkLabel(value) {
|
|
25
27
|
const normalized = nonEmpty$1(value, "benchmark label").toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g, "");
|
|
@@ -706,7 +708,12 @@ const usageSchema = z.strictObject({
|
|
|
706
708
|
calls: nonNegativeInteger.nullable(),
|
|
707
709
|
tokens: tokenUsageSchema.nullable(),
|
|
708
710
|
cost: costSchema,
|
|
709
|
-
knownCostUsd: nonNegativeNumber.optional()
|
|
711
|
+
knownCostUsd: nonNegativeNumber.optional(),
|
|
712
|
+
partialTokens: z.strictObject({
|
|
713
|
+
input: nonNegativeInteger.nullable(),
|
|
714
|
+
output: nonNegativeInteger.nullable()
|
|
715
|
+
}).optional(),
|
|
716
|
+
tokensEstimated: z.boolean().optional()
|
|
710
717
|
});
|
|
711
718
|
const findingScoreSchema = z.strictObject({
|
|
712
719
|
expectedIssueCount: nonNegativeInteger,
|
|
@@ -1200,7 +1207,7 @@ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
|
|
|
1200
1207
|
"package.json",
|
|
1201
1208
|
"pnpm-lock.yaml"
|
|
1202
1209
|
]);
|
|
1203
|
-
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
1210
|
+
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "4785a1e0d785431f287362808dae777cf73125c551c18260117a711a5f2a6263";
|
|
1204
1211
|
/** The published benchmark evidence was produced at this package version, by
|
|
1205
1212
|
* the retired one-shot direct runner, before trace analysts moved to the
|
|
1206
1213
|
* recursive DSPy RLM engine. Both evidence digests below are historical facts
|
|
@@ -1239,6 +1246,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1239
1246
|
"src/analyst/benchmark-real-model.ts",
|
|
1240
1247
|
"src/analyst/benchmark-report.ts",
|
|
1241
1248
|
"src/analyst/benchmark-response-cache.ts",
|
|
1249
|
+
"src/analyst/benchmark-runner-prime.ts",
|
|
1242
1250
|
"src/analyst/benchmark-scoring.ts",
|
|
1243
1251
|
"src/analyst/benchmark-summary.ts",
|
|
1244
1252
|
"src/analyst/benchmark-verification-artifacts.ts",
|
|
@@ -1251,6 +1259,8 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1251
1259
|
"src/analyst/finding-subject.ts",
|
|
1252
1260
|
"src/analyst/kind-factory.ts",
|
|
1253
1261
|
"src/analyst/parse-tolerant.ts",
|
|
1262
|
+
"src/analyst/prime-bridge-transport.ts",
|
|
1263
|
+
"src/analyst/prime-protocol.ts",
|
|
1254
1264
|
"src/analyst/tool-groups.ts",
|
|
1255
1265
|
"src/analyst/trace-tool-callback.ts",
|
|
1256
1266
|
"src/analyst/types.ts",
|
|
@@ -1299,7 +1309,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1299
1309
|
"src/trace/raw-provider-sink.ts",
|
|
1300
1310
|
"src/verdict-cache.ts"
|
|
1301
1311
|
]);
|
|
1302
|
-
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
1312
|
+
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "46af7a98d6df73d845394827283390bac7a2fb9712f418b14b1093f19a21faea";
|
|
1303
1313
|
function analystBenchmarkImplementationDigest() {
|
|
1304
1314
|
return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256;
|
|
1305
1315
|
}
|
|
@@ -2437,7 +2447,7 @@ function createLocalRunReceipt(config, paths) {
|
|
|
2437
2447
|
traceDir: resolve(config.traceDir),
|
|
2438
2448
|
...config.artifactDir ? { artifactDir: resolve(config.artifactDir) } : {},
|
|
2439
2449
|
outputDir: paths.directory,
|
|
2440
|
-
modelOwnerModule: config.modelOwnerModule
|
|
2450
|
+
...config.modelOwnerModule === void 0 ? {} : { modelOwnerModule: config.modelOwnerModule }
|
|
2441
2451
|
},
|
|
2442
2452
|
command: config.command,
|
|
2443
2453
|
environment: {
|
|
@@ -3596,7 +3606,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
3596
3606
|
const maxModelRequestBytes = config.maxModelRequestBytes ?? 16 * 1024 * 1024;
|
|
3597
3607
|
const maxModelResponseBytes = config.maxModelResponseBytes ?? 4 * 1024 * 1024;
|
|
3598
3608
|
const modelRequestTimeoutMs = config.modelRequestTimeoutMs ?? timeoutMs;
|
|
3599
|
-
const pricing = config.pricing ?? pricingForModel$
|
|
3609
|
+
const pricing = config.pricing ?? pricingForModel$2(model);
|
|
3600
3610
|
const maxCostUsd = config.maxCostUsdPerAnalysis ?? 1;
|
|
3601
3611
|
const costLedger = config.costLedger ?? new CostLedger();
|
|
3602
3612
|
const durability = config.durability ? {
|
|
@@ -3608,7 +3618,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
3608
3618
|
return {
|
|
3609
3619
|
id: "direct",
|
|
3610
3620
|
async analyze(input, context) {
|
|
3611
|
-
const trajectoryId = trajectoryIdFromCaseId$
|
|
3621
|
+
const trajectoryId = trajectoryIdFromCaseId$2(dataset, context.caseId);
|
|
3612
3622
|
const costTags = {
|
|
3613
3623
|
analystId: actor,
|
|
3614
3624
|
benchmarkCaseId: context.caseId,
|
|
@@ -3920,7 +3930,7 @@ function costReceiptMetadata(receipt) {
|
|
|
3920
3930
|
estimatedCostUsd: null
|
|
3921
3931
|
};
|
|
3922
3932
|
}
|
|
3923
|
-
function pricingForModel$
|
|
3933
|
+
function pricingForModel$2(model) {
|
|
3924
3934
|
const pricing = resolveModelPricing(model);
|
|
3925
3935
|
if (!pricing) throw new Error(`no pricing is configured for '${model}'; provide PublicAnalystBenchmarkModelConfig.pricing`);
|
|
3926
3936
|
return {
|
|
@@ -4096,7 +4106,7 @@ async function prepareSingleTraceContext(store, context) {
|
|
|
4096
4106
|
});
|
|
4097
4107
|
}
|
|
4098
4108
|
}
|
|
4099
|
-
function trajectoryIdFromCaseId$
|
|
4109
|
+
function trajectoryIdFromCaseId$2(dataset, caseId) {
|
|
4100
4110
|
const prefix = dataset === "agentrx" ? "agentrx:" : "codetrace:";
|
|
4101
4111
|
if (!caseId.startsWith(prefix) || caseId.length === prefix.length) throw new Error(`unexpected ${dataset} benchmark case id '${caseId}'`);
|
|
4102
4112
|
return caseId.slice(prefix.length);
|
|
@@ -4245,7 +4255,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4245
4255
|
maxToolCalls: config.dspyRlm?.maxToolCalls ?? 80,
|
|
4246
4256
|
maxOutputChars: config.dspyRlm?.maxOutputChars ?? 8e3
|
|
4247
4257
|
};
|
|
4248
|
-
const pricing = config.pricing ?? pricingForModel(config.model);
|
|
4258
|
+
const pricing = config.pricing ?? pricingForModel$1(config.model);
|
|
4249
4259
|
const engine = createDspyRlmTraceEngine({
|
|
4250
4260
|
call: config.call,
|
|
4251
4261
|
callRef: config.callRef,
|
|
@@ -4278,7 +4288,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
4278
4288
|
return {
|
|
4279
4289
|
id: "dspy-rlm",
|
|
4280
4290
|
async analyze(input, context) {
|
|
4281
|
-
const trajectoryId = trajectoryIdFromCaseId(dataset, context.caseId);
|
|
4291
|
+
const trajectoryId = trajectoryIdFromCaseId$1(dataset, context.caseId);
|
|
4282
4292
|
const tags = {
|
|
4283
4293
|
benchmarkCaseId: context.caseId,
|
|
4284
4294
|
benchmarkRepetition: String(context.repetition)
|
|
@@ -4518,7 +4528,7 @@ function publicBenchmarkDefinition(dataset, limits, instructions) {
|
|
|
4518
4528
|
limits
|
|
4519
4529
|
};
|
|
4520
4530
|
}
|
|
4521
|
-
function pricingForModel(model) {
|
|
4531
|
+
function pricingForModel$1(model) {
|
|
4522
4532
|
const pricing = resolveModelPricing(model);
|
|
4523
4533
|
if (!pricing) throw new Error(`no pricing is configured for '${model}'; provide PublicAnalystBenchmarkModelConfig.pricing`);
|
|
4524
4534
|
return {
|
|
@@ -4526,7 +4536,7 @@ function pricingForModel(model) {
|
|
|
4526
4536
|
outputUsdPerMillion: pricing.output * 1e3
|
|
4527
4537
|
};
|
|
4528
4538
|
}
|
|
4529
|
-
function trajectoryIdFromCaseId(dataset, caseId) {
|
|
4539
|
+
function trajectoryIdFromCaseId$1(dataset, caseId) {
|
|
4530
4540
|
const prefix = dataset === "agentrx" ? "agentrx:" : "codetrace:";
|
|
4531
4541
|
if (!caseId.startsWith(prefix) || caseId.length === prefix.length) throw new Error(`unexpected ${dataset} benchmark case id '${caseId}'`);
|
|
4532
4542
|
return caseId.slice(prefix.length);
|
|
@@ -4843,6 +4853,832 @@ function rootAgent(row) {
|
|
|
4843
4853
|
return scalarDistributionValue(row.failures.find((failure) => String(failure.failure_id) === String(rootCauseId))?.failed_agent);
|
|
4844
4854
|
}
|
|
4845
4855
|
//#endregion
|
|
4856
|
+
//#region src/analyst/prime-bridge-transport.ts
|
|
4857
|
+
/**
|
|
4858
|
+
* Default transport on node:http/node:https rather than fetch: undici's fixed
|
|
4859
|
+
* response-header timeout kills prime calls that legitimately run past five
|
|
4860
|
+
* minutes, so the request's AbortSignal is the only deadline.
|
|
4861
|
+
*/
|
|
4862
|
+
function nodeHttpPrimeBridgeTransport() {
|
|
4863
|
+
return ({ url, body, signal }) => {
|
|
4864
|
+
const target = new URL(url);
|
|
4865
|
+
if (target.protocol !== "http:" && target.protocol !== "https:") throw new TypeError(`bridge URL must be http: or https:, got ${target.protocol}`);
|
|
4866
|
+
const send = target.protocol === "https:" ? request$1 : request;
|
|
4867
|
+
const encoded = JSON.stringify(body);
|
|
4868
|
+
return new Promise((resolvePromise, rejectPromise) => {
|
|
4869
|
+
const req = send({
|
|
4870
|
+
hostname: target.hostname,
|
|
4871
|
+
port: target.port,
|
|
4872
|
+
path: `${target.pathname}${target.search}`,
|
|
4873
|
+
method: "POST",
|
|
4874
|
+
headers: {
|
|
4875
|
+
"content-type": "application/json",
|
|
4876
|
+
"content-length": Buffer.byteLength(encoded)
|
|
4877
|
+
},
|
|
4878
|
+
signal
|
|
4879
|
+
}, (res) => {
|
|
4880
|
+
const chunks = [];
|
|
4881
|
+
res.on("data", (chunk) => chunks.push(chunk));
|
|
4882
|
+
res.on("end", () => resolvePromise({
|
|
4883
|
+
status: res.statusCode ?? 0,
|
|
4884
|
+
text: Buffer.concat(chunks).toString("utf8")
|
|
4885
|
+
}));
|
|
4886
|
+
res.on("error", rejectPromise);
|
|
4887
|
+
});
|
|
4888
|
+
req.on("error", rejectPromise);
|
|
4889
|
+
req.end(encoded);
|
|
4890
|
+
});
|
|
4891
|
+
};
|
|
4892
|
+
}
|
|
4893
|
+
//#endregion
|
|
4894
|
+
//#region src/analyst/prime-protocol.ts
|
|
4895
|
+
function buildPrimePrompt(spec) {
|
|
4896
|
+
return [
|
|
4897
|
+
`QUESTION: ${spec.question}`,
|
|
4898
|
+
"",
|
|
4899
|
+
...spec.taskDefinition === void 0 ? [] : [
|
|
4900
|
+
"TASK DEFINITION:",
|
|
4901
|
+
spec.taskDefinition,
|
|
4902
|
+
""
|
|
4903
|
+
],
|
|
4904
|
+
...spec.contractLines,
|
|
4905
|
+
"",
|
|
4906
|
+
spec.trajectoryHeader,
|
|
4907
|
+
spec.renderedTrajectory,
|
|
4908
|
+
...spec.trailer === void 0 ? [] : ["", spec.trailer]
|
|
4909
|
+
].join("\n");
|
|
4910
|
+
}
|
|
4911
|
+
/** Carries the malformed reply and the contract — never the trajectory. */
|
|
4912
|
+
function buildPrimeRepairPrompt(spec) {
|
|
4913
|
+
return [
|
|
4914
|
+
"Your previous reply to a trace-analysis task was structurally malformed and could not be parsed",
|
|
4915
|
+
`(${spec.defect}). Below is your previous reply verbatim. Re-emit ONLY the corrected JSON — one`,
|
|
4916
|
+
"fenced ```json block, no other text, no tools. The JSON object has exactly two fields:",
|
|
4917
|
+
...spec.repairContractLines,
|
|
4918
|
+
"",
|
|
4919
|
+
"PREVIOUS REPLY:",
|
|
4920
|
+
spec.previousReply
|
|
4921
|
+
].join("\n");
|
|
4922
|
+
}
|
|
4923
|
+
/**
|
|
4924
|
+
* Recover the reply's JSON object.
|
|
4925
|
+
*
|
|
4926
|
+
* Distinct from `extractJsonPayload` in ../llm-client, which serves a response
|
|
4927
|
+
* that DECLARES a JSON root and therefore must not scan onward. A prime reply
|
|
4928
|
+
* is prose plus a fenced block, and when the model emits several fences the
|
|
4929
|
+
* last one is its answer — so fences are scanned in reverse, and only then is a
|
|
4930
|
+
* brace-to-brace slice tried.
|
|
4931
|
+
*/
|
|
4932
|
+
function extractPrimeJsonObject(text) {
|
|
4933
|
+
const direct = parsePrimeJsonObject(text);
|
|
4934
|
+
if (direct) return direct;
|
|
4935
|
+
const fenced = [...text.matchAll(/```(?:json)?\s*\n?([\s\S]*?)```/g)];
|
|
4936
|
+
for (let index = fenced.length - 1; index >= 0; index -= 1) {
|
|
4937
|
+
const candidate = parsePrimeJsonObject(fenced[index][1]);
|
|
4938
|
+
if (candidate) return candidate;
|
|
4939
|
+
}
|
|
4940
|
+
const start = text.indexOf("{");
|
|
4941
|
+
const end = text.lastIndexOf("}");
|
|
4942
|
+
if (start >= 0 && end > start) {
|
|
4943
|
+
const candidate = parsePrimeJsonObject(text.slice(start, end + 1));
|
|
4944
|
+
if (candidate) return candidate;
|
|
4945
|
+
}
|
|
4946
|
+
return null;
|
|
4947
|
+
}
|
|
4948
|
+
/** Why the reply cannot be read as a prime answer, or null when it can. */
|
|
4949
|
+
function primeReplyDefect(parsed, rowsField) {
|
|
4950
|
+
if (parsed === null) return "no parseable JSON object";
|
|
4951
|
+
if (!Array.isArray(parsed[rowsField])) return `JSON has no "${rowsField}" array`;
|
|
4952
|
+
return null;
|
|
4953
|
+
}
|
|
4954
|
+
function parsePrimeJsonObject(text) {
|
|
4955
|
+
try {
|
|
4956
|
+
const value = JSON.parse(text.trim());
|
|
4957
|
+
return typeof value === "object" && value !== null && !Array.isArray(value) ? value : null;
|
|
4958
|
+
} catch {
|
|
4959
|
+
return null;
|
|
4960
|
+
}
|
|
4961
|
+
}
|
|
4962
|
+
function emptyPrimeRawUsage() {
|
|
4963
|
+
return {
|
|
4964
|
+
calls: null,
|
|
4965
|
+
inputTokens: null,
|
|
4966
|
+
outputTokens: null,
|
|
4967
|
+
bridgeEstimated: false
|
|
4968
|
+
};
|
|
4969
|
+
}
|
|
4970
|
+
/** Read the bridge's OpenAI-shaped `usage` object. */
|
|
4971
|
+
function normalizePrimeUsage(raw) {
|
|
4972
|
+
if (typeof raw !== "object" || raw === null) return emptyPrimeRawUsage();
|
|
4973
|
+
const record = raw;
|
|
4974
|
+
return {
|
|
4975
|
+
calls: tokenCountOrNull(record.model_requests),
|
|
4976
|
+
inputTokens: tokenCountOrNull(record.prompt_tokens),
|
|
4977
|
+
outputTokens: tokenCountOrNull(record.completion_tokens),
|
|
4978
|
+
bridgeEstimated: record.estimated === true
|
|
4979
|
+
};
|
|
4980
|
+
}
|
|
4981
|
+
/**
|
|
4982
|
+
* Sum two turns. Each side poisons independently: two turns that both report
|
|
4983
|
+
* input and neither report output yield a real input total beside a null
|
|
4984
|
+
* output, because discarding a measured count is as wrong as inventing one.
|
|
4985
|
+
*/
|
|
4986
|
+
function mergePrimeRawUsage(a, b) {
|
|
4987
|
+
return {
|
|
4988
|
+
calls: sumOrNull(a.calls, b.calls),
|
|
4989
|
+
inputTokens: sumOrNull(a.inputTokens, b.inputTokens),
|
|
4990
|
+
outputTokens: sumOrNull(a.outputTokens, b.outputTokens),
|
|
4991
|
+
bridgeEstimated: a.bridgeEstimated || b.bridgeEstimated
|
|
4992
|
+
};
|
|
4993
|
+
}
|
|
4994
|
+
function sumOrNull(a, b) {
|
|
4995
|
+
return a !== null && b !== null ? a + b : null;
|
|
4996
|
+
}
|
|
4997
|
+
function tokenCountOrNull(value) {
|
|
4998
|
+
return typeof value === "number" && Number.isSafeInteger(value) && value >= 0 ? value : null;
|
|
4999
|
+
}
|
|
5000
|
+
/**
|
|
5001
|
+
* Bind raw prime usage to agent-eval's typed receipt.
|
|
5002
|
+
*
|
|
5003
|
+
* `RunTokenUsage.input` and `.output` are non-nullable, so a one-sided count
|
|
5004
|
+
* cannot round-trip through `tokens` without writing a zero nobody measured.
|
|
5005
|
+
* The complete-accounting field therefore stays null, the reported side is
|
|
5006
|
+
* carried verbatim in `partialTokens`, and its price becomes the receipt's
|
|
5007
|
+
* `knownCostUsd` lower bound.
|
|
5008
|
+
*
|
|
5009
|
+
* Only agent-eval calls this; consumers with no pricing table read
|
|
5010
|
+
* `PrimeRawUsage` directly.
|
|
5011
|
+
*/
|
|
5012
|
+
function analystUsageReceiptFromPrimeUsage(usage, pricing) {
|
|
5013
|
+
const { calls, inputTokens, outputTokens, bridgeEstimated } = usage;
|
|
5014
|
+
const estimatedTokens = bridgeEstimated ? { tokensEstimated: true } : {};
|
|
5015
|
+
if (inputTokens !== null && outputTokens !== null) return {
|
|
5016
|
+
calls,
|
|
5017
|
+
tokens: {
|
|
5018
|
+
input: inputTokens,
|
|
5019
|
+
output: outputTokens
|
|
5020
|
+
},
|
|
5021
|
+
cost: {
|
|
5022
|
+
kind: "estimated",
|
|
5023
|
+
usd: priceTokens(inputTokens, outputTokens, pricing)
|
|
5024
|
+
},
|
|
5025
|
+
...estimatedTokens
|
|
5026
|
+
};
|
|
5027
|
+
if (inputTokens === null && outputTokens === null) return {
|
|
5028
|
+
calls,
|
|
5029
|
+
tokens: null,
|
|
5030
|
+
cost: {
|
|
5031
|
+
kind: "uncaptured",
|
|
5032
|
+
usd: null
|
|
5033
|
+
},
|
|
5034
|
+
...estimatedTokens
|
|
5035
|
+
};
|
|
5036
|
+
return {
|
|
5037
|
+
calls,
|
|
5038
|
+
tokens: null,
|
|
5039
|
+
partialTokens: {
|
|
5040
|
+
input: inputTokens,
|
|
5041
|
+
output: outputTokens
|
|
5042
|
+
},
|
|
5043
|
+
cost: {
|
|
5044
|
+
kind: "uncaptured",
|
|
5045
|
+
usd: null
|
|
5046
|
+
},
|
|
5047
|
+
knownCostUsd: priceTokens(inputTokens ?? 0, outputTokens ?? 0, pricing),
|
|
5048
|
+
...estimatedTokens
|
|
5049
|
+
};
|
|
5050
|
+
}
|
|
5051
|
+
function priceTokens(input, output, pricing) {
|
|
5052
|
+
return (input * pricing.inputUsdPerMillion + output * pricing.outputUsdPerMillion) / 1e6;
|
|
5053
|
+
}
|
|
5054
|
+
/**
|
|
5055
|
+
* Run the protocol: one call, one bounded repair turn on a structurally
|
|
5056
|
+
* malformed reply, then decode. Zero valid rows from a well-formed reply is an
|
|
5057
|
+
* honest null, not a failure.
|
|
5058
|
+
*/
|
|
5059
|
+
async function runPrimeExchange(options) {
|
|
5060
|
+
const { contract } = options;
|
|
5061
|
+
const turns = [];
|
|
5062
|
+
const repair = {
|
|
5063
|
+
attempted: false,
|
|
5064
|
+
succeeded: null
|
|
5065
|
+
};
|
|
5066
|
+
const first = await callPrimeTurn(options, options.prompt);
|
|
5067
|
+
if (!first.ok) return {
|
|
5068
|
+
ok: false,
|
|
5069
|
+
failure: first.failure,
|
|
5070
|
+
usage: mergeTurns(turns),
|
|
5071
|
+
turns,
|
|
5072
|
+
repair
|
|
5073
|
+
};
|
|
5074
|
+
turns.push({
|
|
5075
|
+
turn: "first",
|
|
5076
|
+
usage: first.usage,
|
|
5077
|
+
rawUsage: first.rawUsage
|
|
5078
|
+
});
|
|
5079
|
+
let reply = first.content;
|
|
5080
|
+
let parsed = extractPrimeJsonObject(reply);
|
|
5081
|
+
let defect = primeReplyDefect(parsed, contract.rowsField);
|
|
5082
|
+
if (defect !== null && options.repair) {
|
|
5083
|
+
repair.attempted = true;
|
|
5084
|
+
const second = await callPrimeTurn(options, buildPrimeRepairPrompt({
|
|
5085
|
+
defect,
|
|
5086
|
+
previousReply: reply,
|
|
5087
|
+
repairContractLines: contract.repairContractLines
|
|
5088
|
+
}));
|
|
5089
|
+
if (!second.ok) return {
|
|
5090
|
+
ok: false,
|
|
5091
|
+
failure: second.failure,
|
|
5092
|
+
usage: mergeTurns(turns),
|
|
5093
|
+
turns,
|
|
5094
|
+
repair,
|
|
5095
|
+
reply
|
|
5096
|
+
};
|
|
5097
|
+
turns.push({
|
|
5098
|
+
turn: "repair",
|
|
5099
|
+
usage: second.usage,
|
|
5100
|
+
rawUsage: second.rawUsage
|
|
5101
|
+
});
|
|
5102
|
+
reply = second.content;
|
|
5103
|
+
parsed = extractPrimeJsonObject(reply);
|
|
5104
|
+
defect = primeReplyDefect(parsed, contract.rowsField);
|
|
5105
|
+
repair.succeeded = defect === null;
|
|
5106
|
+
}
|
|
5107
|
+
const usage = mergeTurns(turns);
|
|
5108
|
+
if (defect !== null) return {
|
|
5109
|
+
ok: false,
|
|
5110
|
+
failure: {
|
|
5111
|
+
kind: "malformed-reply",
|
|
5112
|
+
message: `${defect} in prime reply${repair.attempted ? " even after the bounded repair turn" : ""}`
|
|
5113
|
+
},
|
|
5114
|
+
usage,
|
|
5115
|
+
turns,
|
|
5116
|
+
repair,
|
|
5117
|
+
reply
|
|
5118
|
+
};
|
|
5119
|
+
const rawRows = parsed[contract.rowsField];
|
|
5120
|
+
const rows = [];
|
|
5121
|
+
const rejected = [];
|
|
5122
|
+
let overflow = 0;
|
|
5123
|
+
rawRows.forEach((row, index) => {
|
|
5124
|
+
const decoded = contract.decodeRow(row, index);
|
|
5125
|
+
if (!decoded.ok) {
|
|
5126
|
+
rejected.push({
|
|
5127
|
+
index,
|
|
5128
|
+
reason: decoded.reason
|
|
5129
|
+
});
|
|
5130
|
+
return;
|
|
5131
|
+
}
|
|
5132
|
+
if (contract.maxRows !== void 0 && rows.length >= contract.maxRows) {
|
|
5133
|
+
overflow += 1;
|
|
5134
|
+
return;
|
|
5135
|
+
}
|
|
5136
|
+
rows.push(decoded.row);
|
|
5137
|
+
});
|
|
5138
|
+
const answer = parsed.answer;
|
|
5139
|
+
return {
|
|
5140
|
+
ok: true,
|
|
5141
|
+
answer: typeof answer === "string" ? answer : null,
|
|
5142
|
+
rows,
|
|
5143
|
+
rejected,
|
|
5144
|
+
reportedRows: rawRows.length,
|
|
5145
|
+
overflow,
|
|
5146
|
+
usage,
|
|
5147
|
+
turns,
|
|
5148
|
+
repair,
|
|
5149
|
+
reply
|
|
5150
|
+
};
|
|
5151
|
+
}
|
|
5152
|
+
async function callPrimeTurn(options, content) {
|
|
5153
|
+
const { transport, url, model, timeoutMs, signal } = options;
|
|
5154
|
+
const controller = new AbortController();
|
|
5155
|
+
const forwardAbort = () => controller.abort(signal?.reason);
|
|
5156
|
+
if (signal?.aborted) controller.abort(signal.reason);
|
|
5157
|
+
else signal?.addEventListener("abort", forwardAbort, { once: true });
|
|
5158
|
+
const deadline = setTimeout(() => controller.abort(), timeoutMs);
|
|
5159
|
+
let result;
|
|
5160
|
+
try {
|
|
5161
|
+
result = await transport({
|
|
5162
|
+
url,
|
|
5163
|
+
body: {
|
|
5164
|
+
model,
|
|
5165
|
+
messages: [{
|
|
5166
|
+
role: "user",
|
|
5167
|
+
content
|
|
5168
|
+
}]
|
|
5169
|
+
},
|
|
5170
|
+
signal: controller.signal
|
|
5171
|
+
});
|
|
5172
|
+
} catch (error) {
|
|
5173
|
+
if (signal?.aborted) return {
|
|
5174
|
+
ok: false,
|
|
5175
|
+
failure: {
|
|
5176
|
+
kind: "aborted",
|
|
5177
|
+
message: "prime exchange cancelled by the caller",
|
|
5178
|
+
cause: error
|
|
5179
|
+
}
|
|
5180
|
+
};
|
|
5181
|
+
if (controller.signal.aborted) return {
|
|
5182
|
+
ok: false,
|
|
5183
|
+
failure: {
|
|
5184
|
+
kind: "deadline",
|
|
5185
|
+
message: `bridge call exceeded ${timeoutMs}ms`
|
|
5186
|
+
}
|
|
5187
|
+
};
|
|
5188
|
+
return {
|
|
5189
|
+
ok: false,
|
|
5190
|
+
failure: {
|
|
5191
|
+
kind: "transport",
|
|
5192
|
+
message: `bridge transport failure: ${error instanceof Error ? error.message : String(error)}`
|
|
5193
|
+
}
|
|
5194
|
+
};
|
|
5195
|
+
} finally {
|
|
5196
|
+
clearTimeout(deadline);
|
|
5197
|
+
signal?.removeEventListener("abort", forwardAbort);
|
|
5198
|
+
}
|
|
5199
|
+
if (result.status !== 200) {
|
|
5200
|
+
const bodySnippet = result.text.slice(0, 500);
|
|
5201
|
+
return {
|
|
5202
|
+
ok: false,
|
|
5203
|
+
failure: {
|
|
5204
|
+
kind: "http-status",
|
|
5205
|
+
message: `bridge HTTP ${result.status}: ${bodySnippet}`,
|
|
5206
|
+
status: result.status,
|
|
5207
|
+
bodySnippet
|
|
5208
|
+
}
|
|
5209
|
+
};
|
|
5210
|
+
}
|
|
5211
|
+
let response;
|
|
5212
|
+
try {
|
|
5213
|
+
response = JSON.parse(result.text);
|
|
5214
|
+
} catch {
|
|
5215
|
+
return {
|
|
5216
|
+
ok: false,
|
|
5217
|
+
failure: {
|
|
5218
|
+
kind: "unparseable-json",
|
|
5219
|
+
message: `bridge returned unparseable JSON (${result.text.length} bytes)`
|
|
5220
|
+
}
|
|
5221
|
+
};
|
|
5222
|
+
}
|
|
5223
|
+
const replyContent = primeReplyContent(response);
|
|
5224
|
+
if (replyContent === null) return {
|
|
5225
|
+
ok: false,
|
|
5226
|
+
failure: {
|
|
5227
|
+
kind: "no-content",
|
|
5228
|
+
message: "bridge reply carries no message content"
|
|
5229
|
+
}
|
|
5230
|
+
};
|
|
5231
|
+
const rawUsage = primeReplyUsage(response);
|
|
5232
|
+
return {
|
|
5233
|
+
ok: true,
|
|
5234
|
+
content: replyContent,
|
|
5235
|
+
usage: normalizePrimeUsage(rawUsage),
|
|
5236
|
+
rawUsage
|
|
5237
|
+
};
|
|
5238
|
+
}
|
|
5239
|
+
/**
|
|
5240
|
+
* Fold from the FIRST turn, never from an empty receipt: an all-null identity
|
|
5241
|
+
* would poison every side it merged with and erase counts the bridge reported.
|
|
5242
|
+
*/
|
|
5243
|
+
function mergeTurns(turns) {
|
|
5244
|
+
if (turns.length === 0) return emptyPrimeRawUsage();
|
|
5245
|
+
return turns.slice(1).reduce((total, turn) => mergePrimeRawUsage(total, turn.usage), turns[0].usage);
|
|
5246
|
+
}
|
|
5247
|
+
function primeReplyContent(response) {
|
|
5248
|
+
if (typeof response !== "object" || response === null) return null;
|
|
5249
|
+
const choices = response.choices;
|
|
5250
|
+
if (!Array.isArray(choices) || choices.length === 0) return null;
|
|
5251
|
+
const message = choices[0]?.message;
|
|
5252
|
+
if (typeof message !== "object" || message === null) return null;
|
|
5253
|
+
const content = message.content;
|
|
5254
|
+
return typeof content === "string" && content.length > 0 ? content : null;
|
|
5255
|
+
}
|
|
5256
|
+
function primeReplyUsage(response) {
|
|
5257
|
+
if (typeof response !== "object" || response === null) return null;
|
|
5258
|
+
return response.usage ?? null;
|
|
5259
|
+
}
|
|
5260
|
+
/**
|
|
5261
|
+
* Render, measure, fall back to the capped projection, re-measure, fail loud.
|
|
5262
|
+
*
|
|
5263
|
+
* Inline is the only delivery prime has, so an oversized trajectory is a
|
|
5264
|
+
* refusal rather than a silent truncation: dropping spans would understate the
|
|
5265
|
+
* trajectory and the analyst would answer a question about a different run.
|
|
5266
|
+
*/
|
|
5267
|
+
async function projectPrimeTrajectory(source, limits) {
|
|
5268
|
+
let fetch = "full";
|
|
5269
|
+
let items = await source.full();
|
|
5270
|
+
if (items === null) {
|
|
5271
|
+
fetch = "capped";
|
|
5272
|
+
items = await source.capped();
|
|
5273
|
+
}
|
|
5274
|
+
let rendered = JSON.stringify(items);
|
|
5275
|
+
if (rendered.length > limits.maxInlineChars && fetch === "full") {
|
|
5276
|
+
fetch = "capped";
|
|
5277
|
+
items = await source.capped();
|
|
5278
|
+
rendered = JSON.stringify(items);
|
|
5279
|
+
}
|
|
5280
|
+
if (rendered.length > limits.maxInlineChars) return {
|
|
5281
|
+
ok: false,
|
|
5282
|
+
reason: `trajectory renders to ${rendered.length} chars even at ${source.cappedDescription}; inline delivery impossible`,
|
|
5283
|
+
renderedChars: rendered.length
|
|
5284
|
+
};
|
|
5285
|
+
return {
|
|
5286
|
+
ok: true,
|
|
5287
|
+
items,
|
|
5288
|
+
rendered,
|
|
5289
|
+
delivery: {
|
|
5290
|
+
mode: "inline-json",
|
|
5291
|
+
fetch,
|
|
5292
|
+
renderedChars: rendered.length
|
|
5293
|
+
}
|
|
5294
|
+
};
|
|
5295
|
+
}
|
|
5296
|
+
/**
|
|
5297
|
+
* Digest of everything a consumer can send to the bridge under the prime
|
|
5298
|
+
* protocol, recorded per observation so a prime result names the exact contract
|
|
5299
|
+
* that produced it.
|
|
5300
|
+
*
|
|
5301
|
+
* Computed over the ACTUALLY composed contract, so two consumers that both
|
|
5302
|
+
* stamp `analyst_id: 'prime'` while asking materially different questions get
|
|
5303
|
+
* different digests by construction. That is what makes 'prime' a reproducible
|
|
5304
|
+
* claim rather than a label.
|
|
5305
|
+
*/
|
|
5306
|
+
function primeProtocolSha256(identity) {
|
|
5307
|
+
return createHash("sha256").update(JSON.stringify({
|
|
5308
|
+
kind: "prime-analyst-protocol",
|
|
5309
|
+
question: identity.question,
|
|
5310
|
+
taskPrompt: identity.taskDefinition ?? null,
|
|
5311
|
+
outputContract: identity.contractLines,
|
|
5312
|
+
repairContract: buildPrimeRepairPrompt({
|
|
5313
|
+
defect: "<defect>",
|
|
5314
|
+
previousReply: "<previous-reply>",
|
|
5315
|
+
repairContractLines: identity.repairContractLines
|
|
5316
|
+
}),
|
|
5317
|
+
limits: identity.limits
|
|
5318
|
+
})).digest("hex");
|
|
5319
|
+
}
|
|
5320
|
+
//#endregion
|
|
5321
|
+
//#region src/analyst/benchmark-runner-prime.ts
|
|
5322
|
+
/**
|
|
5323
|
+
* Prime analyst arm: the RLM coding agent reached through an OpenAI-compatible
|
|
5324
|
+
* cli-bridge solves the CodeTraceBench incorrect-step task as a one-shot trace
|
|
5325
|
+
* analyst.
|
|
5326
|
+
*
|
|
5327
|
+
* The runner consumes the same prepared benchmark cases every other runner
|
|
5328
|
+
* receives — the trace store already carries the appended final-verification
|
|
5329
|
+
* spans — and produces findings through the same published block expansion, so
|
|
5330
|
+
* a prime observation and a dspy-rlm observation differ only in which analyst
|
|
5331
|
+
* produced the blocks.
|
|
5332
|
+
*
|
|
5333
|
+
* The protocol itself — prompt composition, the bounded repair turn, reply
|
|
5334
|
+
* extraction, the projection ladder, usage normalization — lives in
|
|
5335
|
+
* `./prime-protocol`, which knows nothing about CodeTraceBench. This file is
|
|
5336
|
+
* the benchmark's binding to it: the block row grammar, the store-backed
|
|
5337
|
+
* projection source, and the benchmark observation shape.
|
|
5338
|
+
*
|
|
5339
|
+
* Trajectory delivery is inline JSON in the prompt. The dspy typed path binds
|
|
5340
|
+
* the viewTrace span projection as a REPL variable; prime has no REPL, so the
|
|
5341
|
+
* same projection is serialized into the prompt. When the full projection is
|
|
5342
|
+
* oversized the runner falls back to chunked viewSpans over the same
|
|
5343
|
+
* projection surface with a per-attribute byte cap, and fails loud if the
|
|
5344
|
+
* result still exceeds the inline budget.
|
|
5345
|
+
*
|
|
5346
|
+
* A structurally malformed reply gets ONE bounded repair turn (disable with
|
|
5347
|
+
* `repair: false`): a second stateless call carrying the malformed reply plus
|
|
5348
|
+
* the output contract — never the trajectory — mirroring the dspy arm's typed
|
|
5349
|
+
* repair so both arms face the same structured-output affordance. Still
|
|
5350
|
+
* malformed after repair = failed observation with a typed error, exactly how
|
|
5351
|
+
* a dspy-rlm failure is recorded. Zero valid blocks from a well-formed reply
|
|
5352
|
+
* is an honest null, not a failure.
|
|
5353
|
+
*/
|
|
5354
|
+
const PRIME_ANALYST_ID = "prime";
|
|
5355
|
+
const PRIME_QUESTION = "Which assistant steps are incorrect under the CodeTraceBench definition?";
|
|
5356
|
+
/** Ceiling on the serialized trajectory JSON embedded in the prompt. */
|
|
5357
|
+
const MAX_INLINE_TRAJECTORY_CHARS = 36e4;
|
|
5358
|
+
/** Per-attribute projection cap used by the chunked viewSpans fallback. */
|
|
5359
|
+
const CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP = 1200;
|
|
5360
|
+
/** Minimal per-attribute cap used only to enumerate span ids in store order. */
|
|
5361
|
+
const SPAN_ID_ENUMERATION_ATTRIBUTE_BYTE_CAP = 64;
|
|
5362
|
+
/** viewSpans accepts at most 100 ids per call; 40 keeps each response bounded. */
|
|
5363
|
+
const VIEW_SPANS_CHUNK_SIZE = 40;
|
|
5364
|
+
const PRIME_SEVERITIES = /* @__PURE__ */ new Set([
|
|
5365
|
+
"critical",
|
|
5366
|
+
"high",
|
|
5367
|
+
"medium",
|
|
5368
|
+
"low",
|
|
5369
|
+
"info"
|
|
5370
|
+
]);
|
|
5371
|
+
var PrimeBridgeTransportError = class extends Error {};
|
|
5372
|
+
var PrimeBridgeHttpError = class extends Error {
|
|
5373
|
+
status;
|
|
5374
|
+
constructor(status, bodySnippet) {
|
|
5375
|
+
super(`bridge HTTP ${status}: ${bodySnippet}`);
|
|
5376
|
+
this.status = status;
|
|
5377
|
+
}
|
|
5378
|
+
};
|
|
5379
|
+
var PrimeMalformedReplyError = class extends Error {};
|
|
5380
|
+
var PrimeTraceProjectionError = class extends Error {};
|
|
5381
|
+
/**
|
|
5382
|
+
* Short-strings rule: long reply strings get corrupted when the bridge splices
|
|
5383
|
+
* its backend's stream, so the contract forbids a rationale field and caps
|
|
5384
|
+
* every string the model must emit.
|
|
5385
|
+
*/
|
|
5386
|
+
const PRIME_OUTPUT_CONTRACT_LINES = [
|
|
5387
|
+
"OUTPUT CONTRACT (supersedes any transport wording above — you have no trace tools and no REPL):",
|
|
5388
|
+
"You are a one-shot analyst. Every fact you need is in the TRAJECTORY JSON below.",
|
|
5389
|
+
"Do not run shell commands, do not read or write files, do not use any tools.",
|
|
5390
|
+
"Reply with EXACTLY one fenced ```json code block and no other fenced block. The JSON object has exactly two fields:",
|
|
5391
|
+
" \"answer\": string — ONE short sentence (max 300 chars) naming the latest failure evidence you traced from.",
|
|
5392
|
+
" \"blocks\": array (possibly empty) of failure blocks, each exactly:",
|
|
5393
|
+
" {\"first_step\": int, \"last_step\": int, \"consequence_step\": int,",
|
|
5394
|
+
" \"escape_status\": \"escaped\"|\"unescaped\",",
|
|
5395
|
+
" \"severity\": \"critical\"|\"high\"|\"medium\"|\"low\"|\"info\",",
|
|
5396
|
+
" \"claim\": string (ONE short sentence, max 200 chars),",
|
|
5397
|
+
" \"confidence\": number 0..1}",
|
|
5398
|
+
"Do NOT include a rationale field. Keep every string SHORT — long strings get corrupted in transport and void your work.",
|
|
5399
|
+
`Report at most 16 blocks; a block spans at most 12 steps.`,
|
|
5400
|
+
"Every step number must be the n of an existing assistant span with span_id \"step-<n>\" and kind \"LLM\" in the trajectory below; never cite TOOL, CHAIN, or AGENT spans.",
|
|
5401
|
+
"\"blocks\" is [] only for a clean trajectory."
|
|
5402
|
+
];
|
|
5403
|
+
const PRIME_REPAIR_CONTRACT_LINES = [
|
|
5404
|
+
" \"answer\": string (ONE short sentence, max 300 chars)",
|
|
5405
|
+
" \"blocks\": array (possibly empty) of {\"first_step\": int, \"last_step\": int, \"consequence_step\": int,",
|
|
5406
|
+
" \"escape_status\": \"escaped\"|\"unescaped\", \"severity\": \"critical\"|\"high\"|\"medium\"|\"low\"|\"info\",",
|
|
5407
|
+
" \"claim\": string (max 200 chars), \"confidence\": number 0..1}",
|
|
5408
|
+
"No rationale field. Keep every string SHORT. Preserve the step numbers and verdicts of your previous reply exactly; shorten prose freely."
|
|
5409
|
+
];
|
|
5410
|
+
const PRIME_PROTOCOL_IDENTITY = {
|
|
5411
|
+
question: PRIME_QUESTION,
|
|
5412
|
+
taskDefinition: CODE_TRACE_BENCH_ANALYST_PROMPT,
|
|
5413
|
+
contractLines: PRIME_OUTPUT_CONTRACT_LINES,
|
|
5414
|
+
repairContractLines: PRIME_REPAIR_CONTRACT_LINES,
|
|
5415
|
+
limits: {
|
|
5416
|
+
maxBlocks: 16,
|
|
5417
|
+
maxBlockSteps: 12,
|
|
5418
|
+
maxInlineTrajectoryChars: MAX_INLINE_TRAJECTORY_CHARS,
|
|
5419
|
+
chunkedProjectionAttributeByteCap: CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP
|
|
5420
|
+
}
|
|
5421
|
+
};
|
|
5422
|
+
/**
|
|
5423
|
+
* The block row grammar. No `maxRows`: the count cap belongs to
|
|
5424
|
+
* `expandCodeTraceFailureBlocks`, which drops the offending block and names it
|
|
5425
|
+
* in `diagnostics.droppedBlocks`, so capping here would erase that record.
|
|
5426
|
+
*/
|
|
5427
|
+
const PRIME_BLOCK_CONTRACT = {
|
|
5428
|
+
rowsField: "blocks",
|
|
5429
|
+
contractLines: PRIME_OUTPUT_CONTRACT_LINES,
|
|
5430
|
+
repairContractLines: PRIME_REPAIR_CONTRACT_LINES,
|
|
5431
|
+
decodeRow(row) {
|
|
5432
|
+
const reason = blockRowDefect(row);
|
|
5433
|
+
if (reason !== null) return {
|
|
5434
|
+
ok: false,
|
|
5435
|
+
reason
|
|
5436
|
+
};
|
|
5437
|
+
return {
|
|
5438
|
+
ok: true,
|
|
5439
|
+
row: blockFromRow(row)
|
|
5440
|
+
};
|
|
5441
|
+
}
|
|
5442
|
+
};
|
|
5443
|
+
/**
|
|
5444
|
+
* Digest of everything this runner can send to the bridge, recorded per
|
|
5445
|
+
* observation so a prime result names the exact contract that produced it.
|
|
5446
|
+
*/
|
|
5447
|
+
function primeAnalystProtocolSha256() {
|
|
5448
|
+
return primeProtocolSha256(PRIME_PROTOCOL_IDENTITY);
|
|
5449
|
+
}
|
|
5450
|
+
/** CodeTraceBench-only: the prompt and output contract speak its block grammar. */
|
|
5451
|
+
function createPrimeBenchmarkRunner(options) {
|
|
5452
|
+
const baseUrl = requiredString(options.baseUrl, "baseUrl").replace(/\/+$/, "");
|
|
5453
|
+
const model = requiredString(options.model, "model");
|
|
5454
|
+
const timeoutMs = positiveSafeInteger(options.timeoutMs, "timeoutMs");
|
|
5455
|
+
const repair = options.repair;
|
|
5456
|
+
if (typeof repair !== "boolean") throw new TypeError("repair must be a boolean");
|
|
5457
|
+
const pricing = options.pricing ?? pricingForModel(model);
|
|
5458
|
+
const transport = options.transport ?? nodeHttpPrimeBridgeTransport();
|
|
5459
|
+
const url = `${baseUrl}/v1/chat/completions`;
|
|
5460
|
+
return {
|
|
5461
|
+
id: PRIME_ANALYST_ID,
|
|
5462
|
+
async analyze(input, context) {
|
|
5463
|
+
const trajectoryId = trajectoryIdFromCaseId(context.caseId);
|
|
5464
|
+
let usage;
|
|
5465
|
+
let metadata = {
|
|
5466
|
+
analysisMode: "prime-rlm",
|
|
5467
|
+
engine: "prime",
|
|
5468
|
+
bridgeUrl: baseUrl,
|
|
5469
|
+
model,
|
|
5470
|
+
protocolSha256: primeAnalystProtocolSha256()
|
|
5471
|
+
};
|
|
5472
|
+
try {
|
|
5473
|
+
const store = input.traceStore;
|
|
5474
|
+
if (!store) throw new Error("codetracebench prime runner requires a trace store");
|
|
5475
|
+
const projection = await projectPrimeTrajectory(codeTraceProjectionSource(store, trajectoryId, context.signal ? { signal: context.signal } : void 0), { maxInlineChars: MAX_INLINE_TRAJECTORY_CHARS });
|
|
5476
|
+
if (!projection.ok) throw new PrimeTraceProjectionError(projection.reason);
|
|
5477
|
+
const delivery = {
|
|
5478
|
+
mode: projection.delivery.mode,
|
|
5479
|
+
fetch: projection.delivery.fetch === "full" ? "view-trace" : "view-spans-chunked",
|
|
5480
|
+
perAttributeByteCap: projection.delivery.fetch === "full" ? null : CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP,
|
|
5481
|
+
renderedChars: projection.delivery.renderedChars
|
|
5482
|
+
};
|
|
5483
|
+
metadata = {
|
|
5484
|
+
...metadata,
|
|
5485
|
+
delivery
|
|
5486
|
+
};
|
|
5487
|
+
const prompt = buildCodeTracePrompt(trajectoryId, projection.items, projection.rendered);
|
|
5488
|
+
metadata = {
|
|
5489
|
+
...metadata,
|
|
5490
|
+
promptChars: prompt.length
|
|
5491
|
+
};
|
|
5492
|
+
const outcome = await runPrimeExchange({
|
|
5493
|
+
contract: PRIME_BLOCK_CONTRACT,
|
|
5494
|
+
prompt,
|
|
5495
|
+
transport,
|
|
5496
|
+
url,
|
|
5497
|
+
model,
|
|
5498
|
+
timeoutMs,
|
|
5499
|
+
repair,
|
|
5500
|
+
...context.signal ? { signal: context.signal } : {}
|
|
5501
|
+
});
|
|
5502
|
+
if (!outcome.ok && outcome.failure.kind === "aborted") throw abortCause(outcome.failure);
|
|
5503
|
+
if (outcome.turns.length > 0) {
|
|
5504
|
+
usage = analystUsageReceiptFromPrimeUsage(outcome.usage, pricing);
|
|
5505
|
+
metadata = {
|
|
5506
|
+
...metadata,
|
|
5507
|
+
bridgeUsage: bridgeUsageFromTurns(outcome.turns)
|
|
5508
|
+
};
|
|
5509
|
+
}
|
|
5510
|
+
metadata = {
|
|
5511
|
+
...metadata,
|
|
5512
|
+
repair: outcome.repair
|
|
5513
|
+
};
|
|
5514
|
+
if (!outcome.ok) {
|
|
5515
|
+
if (outcome.reply !== void 0) metadata = {
|
|
5516
|
+
...metadata,
|
|
5517
|
+
reply: outcome.reply.slice(0, 4e3)
|
|
5518
|
+
};
|
|
5519
|
+
throw primeFailureError(outcome.failure);
|
|
5520
|
+
}
|
|
5521
|
+
const expanded = await expandCodeTraceFailureBlocks({
|
|
5522
|
+
trajectoryId,
|
|
5523
|
+
blocks: outcome.rows,
|
|
5524
|
+
store,
|
|
5525
|
+
analystId: PRIME_ANALYST_ID,
|
|
5526
|
+
...context.signal ? { signal: context.signal } : {}
|
|
5527
|
+
});
|
|
5528
|
+
return {
|
|
5529
|
+
findings: expanded.findings,
|
|
5530
|
+
usage,
|
|
5531
|
+
metadata: {
|
|
5532
|
+
...metadata,
|
|
5533
|
+
answer: outcome.answer,
|
|
5534
|
+
reportedRows: outcome.reportedRows,
|
|
5535
|
+
rejectedRows: outcome.rejected,
|
|
5536
|
+
blockDiagnostics: expanded.diagnostics
|
|
5537
|
+
}
|
|
5538
|
+
};
|
|
5539
|
+
} catch (error) {
|
|
5540
|
+
if (context.signal?.aborted) throw error;
|
|
5541
|
+
return {
|
|
5542
|
+
findings: [],
|
|
5543
|
+
...usage ? { usage } : {},
|
|
5544
|
+
error: publicBenchmarkError(error, []),
|
|
5545
|
+
metadata
|
|
5546
|
+
};
|
|
5547
|
+
}
|
|
5548
|
+
}
|
|
5549
|
+
};
|
|
5550
|
+
}
|
|
5551
|
+
/**
|
|
5552
|
+
* The trace store, seen through the protocol's two-move projection contract:
|
|
5553
|
+
* the full viewTrace projection, or the chunked viewSpans projection at a
|
|
5554
|
+
* per-attribute byte cap.
|
|
5555
|
+
*/
|
|
5556
|
+
function codeTraceProjectionSource(store, trajectoryId, context) {
|
|
5557
|
+
return {
|
|
5558
|
+
async full() {
|
|
5559
|
+
return (await store.viewTrace({ trace_id: trajectoryId }, context)).spans ?? null;
|
|
5560
|
+
},
|
|
5561
|
+
capped: () => projectSpansChunked(store, trajectoryId, context),
|
|
5562
|
+
cappedDescription: `per-attribute cap ${CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP}`
|
|
5563
|
+
};
|
|
5564
|
+
}
|
|
5565
|
+
/**
|
|
5566
|
+
* Chunked viewSpans projection for traces whose full viewTrace response is
|
|
5567
|
+
* oversized. Span ids come from a minimal-cap viewTrace in store order; every
|
|
5568
|
+
* id must project or the case fails loud — a silently dropped span would
|
|
5569
|
+
* understate the trajectory.
|
|
5570
|
+
*/
|
|
5571
|
+
async function projectSpansChunked(store, trajectoryId, context) {
|
|
5572
|
+
const enumeration = await store.viewTrace({
|
|
5573
|
+
trace_id: trajectoryId,
|
|
5574
|
+
per_attribute_byte_cap: SPAN_ID_ENUMERATION_ATTRIBUTE_BYTE_CAP
|
|
5575
|
+
}, context);
|
|
5576
|
+
if (!enumeration.spans) throw new PrimeTraceProjectionError(`trace '${trajectoryId}' is oversized even at per-attribute cap ${SPAN_ID_ENUMERATION_ATTRIBUTE_BYTE_CAP}; cannot enumerate span ids`);
|
|
5577
|
+
const ids = [];
|
|
5578
|
+
const seen = /* @__PURE__ */ new Set();
|
|
5579
|
+
for (const span of enumeration.spans) if (typeof span.span_id === "string" && span.span_id.length > 0 && !seen.has(span.span_id)) {
|
|
5580
|
+
seen.add(span.span_id);
|
|
5581
|
+
ids.push(span.span_id);
|
|
5582
|
+
}
|
|
5583
|
+
if (ids.length === 0) throw new PrimeTraceProjectionError(`no span ids parsed from trace '${trajectoryId}'`);
|
|
5584
|
+
const projected = [];
|
|
5585
|
+
for (let index = 0; index < ids.length; index += VIEW_SPANS_CHUNK_SIZE) {
|
|
5586
|
+
const chunk = ids.slice(index, index + VIEW_SPANS_CHUNK_SIZE);
|
|
5587
|
+
const result = await store.viewSpans({
|
|
5588
|
+
trace_id: trajectoryId,
|
|
5589
|
+
span_ids: chunk,
|
|
5590
|
+
per_attribute_byte_cap: CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP
|
|
5591
|
+
}, context);
|
|
5592
|
+
if (result.missing_span_ids.length > 0 || result.omitted_span_ids.length > 0 || result.spans.length !== chunk.length) throw new PrimeTraceProjectionError(`viewSpans projected ${result.spans.length}/${chunk.length} spans for chunk at ${index} of '${trajectoryId}'`);
|
|
5593
|
+
projected.push(...result.spans);
|
|
5594
|
+
}
|
|
5595
|
+
return projected;
|
|
5596
|
+
}
|
|
5597
|
+
function buildCodeTracePrompt(trajectoryId, spans, renderedSpans) {
|
|
5598
|
+
const stepSpans = spans.filter((span) => /^step-\d+$/.test(String(span.span_id)));
|
|
5599
|
+
if (stepSpans.length === 0) throw new PrimeTraceProjectionError(`no step-<n> spans in trace '${trajectoryId}'`);
|
|
5600
|
+
const finalVerification = spans.filter(isFinalVerificationSpan);
|
|
5601
|
+
return buildPrimePrompt({
|
|
5602
|
+
question: PRIME_QUESTION,
|
|
5603
|
+
taskDefinition: CODE_TRACE_BENCH_ANALYST_PROMPT,
|
|
5604
|
+
contractLines: PRIME_OUTPUT_CONTRACT_LINES,
|
|
5605
|
+
trajectoryHeader: `TRAJECTORY (trace_id ${trajectoryId}; ${stepSpans.length} assistant step spans; full span projection as JSON):`,
|
|
5606
|
+
renderedTrajectory: renderedSpans,
|
|
5607
|
+
trailer: finalVerification.length > 0 ? `FINAL VERIFICATION SPANS:\n${JSON.stringify(finalVerification)}` : "FINAL VERIFICATION: unavailable for this trajectory — trace backward from the latest failure evidence inside the trajectory itself."
|
|
5608
|
+
});
|
|
5609
|
+
}
|
|
5610
|
+
/** Map the protocol's terminal reason onto this benchmark's typed error classes. */
|
|
5611
|
+
function primeFailureError(failure) {
|
|
5612
|
+
switch (failure.kind) {
|
|
5613
|
+
case "http-status": return new PrimeBridgeHttpError(failure.status, failure.bodySnippet);
|
|
5614
|
+
case "malformed-reply": return new PrimeMalformedReplyError(failure.message);
|
|
5615
|
+
default: return new PrimeBridgeTransportError(failure.message);
|
|
5616
|
+
}
|
|
5617
|
+
}
|
|
5618
|
+
/** A cancelled run is not a result: the caller's error propagates unchanged. */
|
|
5619
|
+
function abortCause(failure) {
|
|
5620
|
+
return failure.cause instanceof Error ? failure.cause : new Error(failure.message);
|
|
5621
|
+
}
|
|
5622
|
+
function bridgeUsageFromTurns(turns) {
|
|
5623
|
+
return {
|
|
5624
|
+
first: turns.find((turn) => turn.turn === "first")?.rawUsage ?? null,
|
|
5625
|
+
repair: turns.find((turn) => turn.turn === "repair")?.rawUsage ?? null
|
|
5626
|
+
};
|
|
5627
|
+
}
|
|
5628
|
+
function isFinalVerificationSpan(span) {
|
|
5629
|
+
if (span.span_id.startsWith("benchmark-verification")) return true;
|
|
5630
|
+
const role = span.attributes["benchmark.evidence.role"];
|
|
5631
|
+
return typeof role === "string" && role.startsWith("final-verification");
|
|
5632
|
+
}
|
|
5633
|
+
function trajectoryIdFromCaseId(caseId) {
|
|
5634
|
+
if (!caseId.startsWith("codetrace:") || caseId.length === 10) throw new Error(`unexpected codetracebench benchmark case id '${caseId}'`);
|
|
5635
|
+
return caseId.slice(10);
|
|
5636
|
+
}
|
|
5637
|
+
function blockRowDefect(row) {
|
|
5638
|
+
if (typeof row !== "object" || row === null || Array.isArray(row)) return "row is not an object";
|
|
5639
|
+
const record = row;
|
|
5640
|
+
for (const field of [
|
|
5641
|
+
"first_step",
|
|
5642
|
+
"last_step",
|
|
5643
|
+
"consequence_step"
|
|
5644
|
+
]) {
|
|
5645
|
+
const value = record[field];
|
|
5646
|
+
if (!Number.isInteger(value) || value < 1) return `${field} must be a positive integer`;
|
|
5647
|
+
}
|
|
5648
|
+
const firstStep = record.first_step;
|
|
5649
|
+
const lastStep = record.last_step;
|
|
5650
|
+
const consequenceStep = record.consequence_step;
|
|
5651
|
+
if (lastStep < firstStep) return "last_step < first_step";
|
|
5652
|
+
if (consequenceStep < firstStep) return "consequence_step < first_step";
|
|
5653
|
+
if (lastStep - firstStep + 1 > 12) return `block spans ${lastStep - firstStep + 1} steps (cap 12)`;
|
|
5654
|
+
if (record.escape_status !== "escaped" && record.escape_status !== "unescaped") return "escape_status must be escaped|unescaped";
|
|
5655
|
+
if (typeof record.severity !== "string" || !PRIME_SEVERITIES.has(record.severity)) return "severity outside the analyst severity enum";
|
|
5656
|
+
if (typeof record.claim !== "string" || record.claim.trim().length === 0 || record.claim.length > 2e3) return "claim must be a 1-2000 char string";
|
|
5657
|
+
if (typeof record.confidence !== "number" || !Number.isFinite(record.confidence) || record.confidence < 0 || record.confidence > 1) return "confidence must be 0..1";
|
|
5658
|
+
return null;
|
|
5659
|
+
}
|
|
5660
|
+
function blockFromRow(row) {
|
|
5661
|
+
const rationale = typeof row.rationale === "string" && row.rationale.trim().length > 0 ? row.rationale.trim().slice(0, 4e3) : void 0;
|
|
5662
|
+
return {
|
|
5663
|
+
firstStep: row.first_step,
|
|
5664
|
+
lastStep: row.last_step,
|
|
5665
|
+
consequenceStep: row.consequence_step,
|
|
5666
|
+
escapeStatus: row.escape_status,
|
|
5667
|
+
severity: row.severity,
|
|
5668
|
+
claim: row.claim.trim(),
|
|
5669
|
+
confidence: row.confidence,
|
|
5670
|
+
...rationale === void 0 ? {} : { rationale }
|
|
5671
|
+
};
|
|
5672
|
+
}
|
|
5673
|
+
function pricingForModel(model) {
|
|
5674
|
+
const pricing = resolveModelPricing(model);
|
|
5675
|
+
if (!pricing) throw new Error(`no pricing is configured for '${model}'; provide PrimeBenchmarkRunnerOptions.pricing`);
|
|
5676
|
+
return {
|
|
5677
|
+
inputUsdPerMillion: pricing.input * 1e3,
|
|
5678
|
+
outputUsdPerMillion: pricing.output * 1e3
|
|
5679
|
+
};
|
|
5680
|
+
}
|
|
5681
|
+
//#endregion
|
|
4846
5682
|
//#region src/analyst/benchmark-report.ts
|
|
4847
5683
|
function renderAnalystBenchmarkMarkdown(result, comparisons = []) {
|
|
4848
5684
|
const { provenance } = result;
|
|
@@ -4973,7 +5809,20 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
|
|
|
4973
5809
|
return benchmarkExitCode(artifact.result, config.analyst);
|
|
4974
5810
|
}
|
|
4975
5811
|
if (await regularFileExists(paths.report)) throw new Error(`benchmark report exists without a completed result; refusing ambiguous resume: ${paths.report}`);
|
|
4976
|
-
const createAnalystRunner = dependencies.createAnalystRunner ?? ((dataset, model) =>
|
|
5812
|
+
const createAnalystRunner = dependencies.createAnalystRunner ?? ((dataset, model) => {
|
|
5813
|
+
if (config.analyst === "prime") {
|
|
5814
|
+
if (!config.prime) throw new Error("analyst 'prime' is missing its bridge configuration");
|
|
5815
|
+
return createPrimeBenchmarkRunner({
|
|
5816
|
+
baseUrl: config.prime.bridgeUrl,
|
|
5817
|
+
model: model.model,
|
|
5818
|
+
timeoutMs: model.timeoutMs,
|
|
5819
|
+
repair: config.prime.repair,
|
|
5820
|
+
...model.pricing ? { pricing: model.pricing } : {}
|
|
5821
|
+
});
|
|
5822
|
+
}
|
|
5823
|
+
const ownerModel = requireModelOwnerSettings(model);
|
|
5824
|
+
return config.analyst === "direct" ? createPublicBenchmarkDirectRunner(dataset, ownerModel) : createPublicBenchmarkRlmRunner(dataset, ownerModel);
|
|
5825
|
+
});
|
|
4977
5826
|
const runners = [emptyPublicBenchmarkRunner(), createAnalystRunner(config.dataset, {
|
|
4978
5827
|
...config.model,
|
|
4979
5828
|
costLedger,
|
|
@@ -5143,21 +5992,31 @@ Run the recursive DSPy trace analyst against public AgentRx or CodeTraceBench la
|
|
|
5143
5992
|
|
|
5144
5993
|
Required:
|
|
5145
5994
|
--dataset agentrx|codetracebench
|
|
5146
|
-
--analyst dspy-rlm|direct
|
|
5995
|
+
--analyst dspy-rlm|direct|prime Scored analyst. Default: dspy-rlm.
|
|
5147
5996
|
'direct' is the one-shot comparison arm.
|
|
5997
|
+
'prime' is the RLM coding agent behind an
|
|
5998
|
+
OpenAI-compatible cli-bridge (codetracebench
|
|
5999
|
+
only; see docs/prime-analyst.md)
|
|
5148
6000
|
--labels <dataset.json|dataset.jsonl>
|
|
5149
6001
|
--trace-dir <one-trace-per-file OTLP JSONL directory>
|
|
5150
6002
|
--artifact-dir <extracted artifact root> Required for CodeTraceBench
|
|
5151
6003
|
--out <new output directory>
|
|
5152
6004
|
--revision <full 40- or 64-character hex digest>
|
|
5153
6005
|
--split <dataset split>
|
|
5154
|
-
--model-owner-module <module> Module exporting
|
|
5155
|
-
the owner keeps
|
|
5156
|
-
|
|
6006
|
+
--model-owner-module <module> dspy-rlm|direct only. Module exporting
|
|
6007
|
+
createModelExecutionOwner; the owner keeps
|
|
6008
|
+
provider credentials and policy
|
|
6009
|
+
--model <provider model id> For prime, the bridge model id in
|
|
6010
|
+
<backend>/<provider>/<model> form, e.g.
|
|
6011
|
+
prime/zai/glm-5.2
|
|
5157
6012
|
--limit <positive case count>
|
|
5158
6013
|
|
|
5159
6014
|
Controls:
|
|
5160
6015
|
--resume Continue an interrupted run in --out
|
|
6016
|
+
--bridge-url <url> prime only. OpenAI-compatible cli-bridge
|
|
6017
|
+
base URL. Default: http://localhost:4181
|
|
6018
|
+
--no-repair prime only. Disable the bounded repair turn
|
|
6019
|
+
for a structurally malformed reply
|
|
5161
6020
|
--seed <integer> Case-selection and comparison seed. Default: 0
|
|
5162
6021
|
--concurrency <positive integer> Parallel benchmark jobs. Default: 1
|
|
5163
6022
|
--repetitions <positive integer> Runs per case and runner. Default: 1
|
|
@@ -5207,7 +6066,17 @@ async function parseCommandConfig(argv, env, dependencies) {
|
|
|
5207
6066
|
const maxCostUsd = positiveFiniteFlag(flags, "max-cost-usd", 5);
|
|
5208
6067
|
const python = flags.get("python")?.trim();
|
|
5209
6068
|
const analyst = flags.get("analyst")?.trim() ?? "dspy-rlm";
|
|
5210
|
-
if (analyst !== "dspy-rlm" && analyst !== "direct") throw new Error("--analyst must be 'dspy-rlm' or '
|
|
6069
|
+
if (analyst !== "dspy-rlm" && analyst !== "direct" && analyst !== "prime") throw new Error("--analyst must be 'dspy-rlm', 'direct', or 'prime'");
|
|
6070
|
+
const bridgeUrl = flags.get("bridge-url")?.trim();
|
|
6071
|
+
if (bridgeUrl !== void 0 && analyst !== "prime") throw new Error("--bridge-url requires --analyst prime");
|
|
6072
|
+
if (bridgeUrl === "") throw new Error("--bridge-url must not be blank");
|
|
6073
|
+
if (flags.has("no-repair") && analyst !== "prime") throw new Error("--no-repair requires --analyst prime");
|
|
6074
|
+
if (analyst === "prime" && dataset !== "codetracebench") throw new Error("--analyst prime requires --dataset codetracebench; the prime runner speaks the CodeTraceBench failure-block contract");
|
|
6075
|
+
if (analyst === "prime" && flags.has("model-owner-module")) throw new Error("--model-owner-module is not used by --analyst prime; the cli-bridge owns model execution");
|
|
6076
|
+
const prime = analyst === "prime" ? {
|
|
6077
|
+
bridgeUrl: bridgeUrl ?? "http://localhost:4181",
|
|
6078
|
+
repair: !flags.has("no-repair")
|
|
6079
|
+
} : void 0;
|
|
5211
6080
|
const rlmSamples = positiveFlag(flags, "rlm-samples", 1);
|
|
5212
6081
|
if (rlmSamples > 1 && analyst !== "dspy-rlm") throw new Error("--rlm-samples above 1 requires --analyst dspy-rlm");
|
|
5213
6082
|
const instructionsFile = flags.get("instructions-file")?.trim();
|
|
@@ -5215,13 +6084,13 @@ async function parseCommandConfig(argv, env, dependencies) {
|
|
|
5215
6084
|
const instructionsOverride = instructionsFile ? readAnalystInstructionsOverride(instructionsFile) : void 0;
|
|
5216
6085
|
if (rlmSamples > 1 && dataset !== "codetracebench") throw new Error("--rlm-samples above 1 requires --dataset codetracebench; step-level consensus is defined on its block grammar");
|
|
5217
6086
|
const model = requiredFlag(flags, "model");
|
|
5218
|
-
const modelOwnerModule = requiredFlag(flags, "model-owner-module");
|
|
5219
|
-
const owner = await (dependencies.loadModelExecutionOwner ?? loadModelExecutionOwner)(modelOwnerModule, {
|
|
6087
|
+
const modelOwnerModule = analyst === "prime" ? void 0 : requiredFlag(flags, "model-owner-module");
|
|
6088
|
+
const owner = modelOwnerModule ? await (dependencies.loadModelExecutionOwner ?? loadModelExecutionOwner)(modelOwnerModule, {
|
|
5220
6089
|
model,
|
|
5221
6090
|
environment: Object.freeze({ ...env })
|
|
5222
|
-
});
|
|
5223
|
-
assertModelExecutionOwner(owner);
|
|
5224
|
-
const pricing = owner
|
|
6091
|
+
}) : void 0;
|
|
6092
|
+
if (owner) assertModelExecutionOwner(owner);
|
|
6093
|
+
const pricing = owner?.pricing ?? benchmarkModelPricing(model);
|
|
5225
6094
|
const maxOutputTokens = positiveFlag(flags, "max-output-tokens", 16384);
|
|
5226
6095
|
const timeoutMs = positiveFlag(flags, "timeout-ms", 3e5);
|
|
5227
6096
|
return {
|
|
@@ -5234,9 +6103,11 @@ async function parseCommandConfig(argv, env, dependencies) {
|
|
|
5234
6103
|
revision: immutableRevision(requiredFlag(flags, "revision")),
|
|
5235
6104
|
split: requiredFlag(flags, "split"),
|
|
5236
6105
|
model: {
|
|
5237
|
-
|
|
5238
|
-
|
|
5239
|
-
|
|
6106
|
+
...owner ? {
|
|
6107
|
+
call: owner.call,
|
|
6108
|
+
callRef: owner.callRef,
|
|
6109
|
+
recordExecution: owner.recordExecution
|
|
6110
|
+
} : { callRef: `cli-bridge:${prime.bridgeUrl}` },
|
|
5240
6111
|
model,
|
|
5241
6112
|
maxOutputTokens,
|
|
5242
6113
|
timeoutMs,
|
|
@@ -5274,11 +6145,22 @@ async function parseCommandConfig(argv, env, dependencies) {
|
|
|
5274
6145
|
rlmSamples,
|
|
5275
6146
|
maxCostUsd,
|
|
5276
6147
|
maxArtifactBytes: positiveFlag(flags, "max-artifact-bytes", DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES),
|
|
5277
|
-
modelOwnerModule,
|
|
6148
|
+
...modelOwnerModule === void 0 ? {} : { modelOwnerModule },
|
|
6149
|
+
...prime === void 0 ? {} : { prime },
|
|
5278
6150
|
command: `agent-eval analyst-benchmark ${argv.filter((argument) => argument !== "--resume").map(shellQuote).join(" ")}`,
|
|
5279
6151
|
resume: flags.has("resume")
|
|
5280
6152
|
};
|
|
5281
6153
|
}
|
|
6154
|
+
/** Fail-loud narrowing: the dspy-rlm and direct analysts require an owner call path. */
|
|
6155
|
+
function requireModelOwnerSettings(model) {
|
|
6156
|
+
const { call, recordExecution } = model;
|
|
6157
|
+
if (typeof call !== "function" || typeof recordExecution !== "function") throw new Error("model-owner execution is required for the dspy-rlm and direct analysts");
|
|
6158
|
+
return {
|
|
6159
|
+
...model,
|
|
6160
|
+
call,
|
|
6161
|
+
recordExecution
|
|
6162
|
+
};
|
|
6163
|
+
}
|
|
5282
6164
|
function parseFlags(argv) {
|
|
5283
6165
|
const flags = /* @__PURE__ */ new Map();
|
|
5284
6166
|
for (let index = 0; index < argv.length; index += 1) {
|
|
@@ -5304,6 +6186,8 @@ const KNOWN_FLAGS = /* @__PURE__ */ new Set([
|
|
|
5304
6186
|
"resume",
|
|
5305
6187
|
"dataset",
|
|
5306
6188
|
"analyst",
|
|
6189
|
+
"bridge-url",
|
|
6190
|
+
"no-repair",
|
|
5307
6191
|
"labels",
|
|
5308
6192
|
"trace-dir",
|
|
5309
6193
|
"artifact-dir",
|
|
@@ -5339,7 +6223,7 @@ const KNOWN_FLAGS = /* @__PURE__ */ new Set([
|
|
|
5339
6223
|
"max-cost-usd",
|
|
5340
6224
|
"max-artifact-bytes"
|
|
5341
6225
|
]);
|
|
5342
|
-
const BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["resume"]);
|
|
6226
|
+
const BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["resume", "no-repair"]);
|
|
5343
6227
|
function assertKnownFlags(flags) {
|
|
5344
6228
|
for (const flag of flags.keys()) if (!KNOWN_FLAGS.has(flag)) throw new Error(`unknown analyst-benchmark flag: --${flag}`);
|
|
5345
6229
|
}
|
|
@@ -5467,6 +6351,6 @@ function shellQuote(value) {
|
|
|
5467
6351
|
return /^[a-zA-Z0-9_./:@%+=,-]+$/.test(value) ? value : `'${value.replaceAll("'", "'\\''")}'`;
|
|
5468
6352
|
}
|
|
5469
6353
|
//#endregion
|
|
5470
|
-
export {
|
|
6354
|
+
export { ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as $, summarizeCodeTraceCalibration as A, publicBenchmarkSystemPrompt as B, createPublicBenchmarkRlmRunner as C, expandCodeTraceFailureBlocks as D, emptyPublicBenchmarkRunner as E, CODE_TRACE_BENCH_ANALYST_PROMPT as F, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as G, appendVerificationArtifactsToOtlp as H, MAX_INCORRECT_BLOCKS as I, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as J, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as K, MAX_INCORRECT_BLOCK_STEPS as L, analystInstructionsOverrideFromText as M, effectiveAnalystProtocolSha256 as N, readAnalystBenchmarkArtifact as O, readAnalystInstructionsOverride as P, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as Q, publicBenchmarkProtocolSha256 as R, selectPublicBenchmarkRows as S, adaptPublicBenchmarkFindings as T, loadCodeTraceVerificationArtifacts as U, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as V, parseVerificationOutcome as W, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as X, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as Y, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as Z, nodeHttpPrimeBridgeTransport as _, primeAnalystProtocolSha256 as a, ANALYST_BENCHMARK_OBSERVATIONS_FILE as at, publicBenchmarkDistributions as b, buildPrimeRepairPrompt as c, summarizeAgentRxCalibration as ct, mergePrimeRawUsage as d, agentRxBenchmarkCase as dt, analystBenchmarkDependencyLockDigest as et, normalizePrimeUsage as f, agentRxPredictionsToFindings as ft, runPrimeExchange as g, projectPrimeTrajectory as h, normalizeBenchmarkLabel as ht, createPrimeBenchmarkRunner as i, ANALYST_BENCHMARK_MANIFEST_FILE as it, compareAnalystRunners as j, renderCodeTraceCalibrationMarkdown as k, emptyPrimeRawUsage as l, codeTraceBenchCase as lt, primeReplyDefect as m, roundAgentRxStep as mt, runAnalystBenchmarkCommand as n, ANALYST_BENCHMARK_COST_LEDGER_FILE as nt, analystUsageReceiptFromPrimeUsage as o, AGENT_RX_UPSTREAM_REVISION as ot, primeProtocolSha256 as p, normalizeAgentRxCategory as pt, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as q, renderAnalystBenchmarkMarkdown as r, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as rt, buildPrimePrompt as s, renderAgentRxCalibrationMarkdown as st, ANALYST_BENCHMARK_HELP as t, analystBenchmarkImplementationDigest as tt, extractPrimeJsonObject as u, codeTracerPredictionsToFindings as ut, loadPublicBenchmarkRows as v, createPublicBenchmarkDirectRunner as w, publicBenchmarkSelectionReport as x, preparePublicAnalystBenchmark as y, publicBenchmarkRlmInstructions as z };
|
|
5471
6355
|
|
|
5472
|
-
//# sourceMappingURL=benchmark-command-
|
|
6356
|
+
//# sourceMappingURL=benchmark-command-CA_NFOmy.js.map
|