@tangle-network/agent-eval 0.144.3 → 0.144.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/dist/analyst/index.d.ts +363 -16
  3. package/dist/analyst/index.d.ts.map +1 -1
  4. package/dist/analyst/index.js +8 -8
  5. package/dist/{benchmark-CYtcIF2V.js → benchmark-B181aMF9.js} +2 -2
  6. package/dist/{benchmark-CYtcIF2V.js.map → benchmark-B181aMF9.js.map} +1 -1
  7. package/dist/{benchmark-CP6kWfj8.d.ts → benchmark-Fmo42QVE.d.ts} +3 -3
  8. package/dist/{benchmark-CP6kWfj8.d.ts.map → benchmark-Fmo42QVE.d.ts.map} +1 -1
  9. package/dist/{benchmark-command-CQPKRUr-.js → benchmark-command-CA_NFOmy.js} +919 -35
  10. package/dist/benchmark-command-CA_NFOmy.js.map +1 -0
  11. package/dist/benchmarks/index.d.ts +1 -1
  12. package/dist/benchmarks/index.js +1 -1
  13. package/dist/{benchmarks-CkG1bWFa.js → benchmarks-CWbsj0t4.js} +2 -2
  14. package/dist/{benchmarks-CkG1bWFa.js.map → benchmarks-CWbsj0t4.js.map} +1 -1
  15. package/dist/campaign/index.d.ts +4 -4
  16. package/dist/campaign/index.js +2 -2
  17. package/dist/{campaign-DjGFyPxH.js → campaign-ClpnD7Ug.js} +2 -2
  18. package/dist/{campaign-DjGFyPxH.js.map → campaign-ClpnD7Ug.js.map} +1 -1
  19. package/dist/cli.js +1 -1
  20. package/dist/{client-Bbht4xxl.d.ts → client-DAb7MWtL.d.ts} +2 -2
  21. package/dist/{client-Bbht4xxl.d.ts.map → client-DAb7MWtL.d.ts.map} +1 -1
  22. package/dist/{completion-verifier-EJERfFwF.d.ts → completion-verifier-VvpHRu78.d.ts} +3 -3
  23. package/dist/{completion-verifier-EJERfFwF.d.ts.map → completion-verifier-VvpHRu78.d.ts.map} +1 -1
  24. package/dist/contract/index.d.ts +6 -6
  25. package/dist/contract/index.js +4 -4
  26. package/dist/control.d.ts +2 -2
  27. package/dist/{default-registry-iXfu2trt.d.ts → default-registry-BwbZ9N9v.d.ts} +5 -5
  28. package/dist/{default-registry-iXfu2trt.d.ts.map → default-registry-BwbZ9N9v.d.ts.map} +1 -1
  29. package/dist/{default-registry-SOyHB6qG.js → default-registry-RLNNoeEP.js} +3 -3
  30. package/dist/{default-registry-SOyHB6qG.js.map → default-registry-RLNNoeEP.js.map} +1 -1
  31. package/dist/{dspy-rlm-engine-IRCG8kdi.js → dspy-rlm-engine-19FQEMBK.js} +2 -2
  32. package/dist/{dspy-rlm-engine-IRCG8kdi.js.map → dspy-rlm-engine-19FQEMBK.js.map} +1 -1
  33. package/dist/{exact-types-BygCBR4L.d.ts → exact-types-BQ7W90C4.d.ts} +2 -2
  34. package/dist/{exact-types-BygCBR4L.d.ts.map → exact-types-BQ7W90C4.d.ts.map} +1 -1
  35. package/dist/{extract-usage-7l1Xq5ti.js → extract-usage-BW27f3XW.js} +2 -2
  36. package/dist/{extract-usage-7l1Xq5ti.js.map → extract-usage-BW27f3XW.js.map} +1 -1
  37. package/dist/{feedback-trajectory-CSIkRLQX.d.ts → feedback-trajectory-WK7x4mhy.d.ts} +3 -3
  38. package/dist/{feedback-trajectory-CSIkRLQX.d.ts.map → feedback-trajectory-WK7x4mhy.d.ts.map} +1 -1
  39. package/dist/hosted/index.d.ts +2 -2
  40. package/dist/{index-BuHs_OnD.d.ts → index-CpxZSlB7.d.ts} +6 -6
  41. package/dist/{index-BuHs_OnD.d.ts.map → index-CpxZSlB7.d.ts.map} +1 -1
  42. package/dist/{index-D_qTihaQ.d.ts → index-CsuAo2-J.d.ts} +4 -4
  43. package/dist/{index-D_qTihaQ.d.ts.map → index-CsuAo2-J.d.ts.map} +1 -1
  44. package/dist/index.d.ts +13 -13
  45. package/dist/index.js +10 -10
  46. package/dist/{kind-factory-Bvwe3pup.js → kind-factory-BHIgPmzS.js} +2 -2
  47. package/dist/{kind-factory-Bvwe3pup.js.map → kind-factory-BHIgPmzS.js.map} +1 -1
  48. package/dist/multishot/index.d.ts +1 -1
  49. package/dist/multishot/index.js +1 -1
  50. package/dist/multishot/index.js.map +1 -1
  51. package/dist/openapi.json +1 -1
  52. package/dist/{replay-DQ-55DC_.d.ts → replay-B7S7Pdbw.d.ts} +3 -3
  53. package/dist/{replay-DQ-55DC_.d.ts.map → replay-B7S7Pdbw.d.ts.map} +1 -1
  54. package/dist/{replay-CqOsGjzU.js → replay-GW61ezMW.js} +4 -4
  55. package/dist/{replay-CqOsGjzU.js.map → replay-GW61ezMW.js.map} +1 -1
  56. package/dist/rl.d.ts +1 -1
  57. package/dist/{run-evidence-H1vRpIdT.d.ts → run-evidence-j5Ynww6L.d.ts} +2 -2
  58. package/dist/{run-evidence-H1vRpIdT.d.ts.map → run-evidence-j5Ynww6L.d.ts.map} +1 -1
  59. package/dist/{semantic-concept-judge-Do5aM9wP.js → semantic-concept-judge-DKRtp2sY.js} +2 -2
  60. package/dist/{semantic-concept-judge-Do5aM9wP.js.map → semantic-concept-judge-DKRtp2sY.js.map} +1 -1
  61. package/dist/{skill-usage-DtpLou9L.d.ts → skill-usage-DUvvudWR.d.ts} +5 -5
  62. package/dist/{skill-usage-DtpLou9L.d.ts.map → skill-usage-DUvvudWR.d.ts.map} +1 -1
  63. package/dist/{skillopt-optimization-method-CvSJGdm3.d.ts → skillopt-optimization-method-CGz9ywhM.d.ts} +4 -4
  64. package/dist/{skillopt-optimization-method-CvSJGdm3.d.ts.map → skillopt-optimization-method-CGz9ywhM.d.ts.map} +1 -1
  65. package/dist/{store-otlp-D4I90_vR.js → store-otlp-CKtTpRhv.js} +2 -2
  66. package/dist/{store-otlp-D4I90_vR.js.map → store-otlp-CKtTpRhv.js.map} +1 -1
  67. package/dist/{tool-groups-CMmsgTzj.d.ts → tool-groups-ByZiqpVk.d.ts} +4 -4
  68. package/dist/{tool-groups-CMmsgTzj.d.ts.map → tool-groups-ByZiqpVk.d.ts.map} +1 -1
  69. package/dist/traces.d.ts +3 -3
  70. package/dist/traces.js +4 -4
  71. package/dist/{types-y8jrxXWd.d.ts → types-CZt1PBIk.d.ts} +19 -1
  72. package/dist/{types-y8jrxXWd.d.ts.map → types-CZt1PBIk.d.ts.map} +1 -1
  73. package/dist/{types-DcJxgsLy.d.ts → types-Dcoaqcsc.d.ts} +2 -2
  74. package/dist/{types-DcJxgsLy.d.ts.map → types-Dcoaqcsc.d.ts.map} +1 -1
  75. package/dist/{usage-receipt-CgxMEBZq.js → usage-receipt-EVI8B8Xu.js} +8 -1
  76. package/dist/usage-receipt-EVI8B8Xu.js.map +1 -0
  77. package/dist/wire/index.d.ts +1 -1
  78. package/docs/prime-analyst.md +120 -0
  79. package/docs/trace-analysis.md +1 -1
  80. package/package.json +3 -3
  81. package/dist/benchmark-command-CQPKRUr-.js.map +0 -1
  82. package/dist/usage-receipt-CgxMEBZq.js.map +0 -1
@@ -2,24 +2,26 @@ import { c as ValidationError, t as AgentEvalError } from "./errors-D-LKuDhb.js"
2
2
  import { s as resolveModelPricing } from "./metrics-C9YY1OcL.js";
3
3
  import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-DMFxsLKr.js";
4
4
  import { c as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-D3EoChAU.js";
5
- import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-Bvwe3pup.js";
6
- import { o as makeFinding, r as usageReceiptFromCostLedger } from "./usage-receipt-CgxMEBZq.js";
5
+ import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-BHIgPmzS.js";
6
+ import { o as makeFinding, r as usageReceiptFromCostLedger } from "./usage-receipt-EVI8B8Xu.js";
7
7
  import { i as hashCanonical, r as canonicalString } from "./canonical-D011XM8r.js";
8
8
  import { M as resolveExternalOptimizerProcessLimits, f as runWithCleanup, n as createRunCostLedger, p as startExternalOptimizerModelProxy, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-D5iN0Xzb.js";
9
- import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-IRCG8kdi.js";
9
+ import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-19FQEMBK.js";
10
10
  import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-DXZIqu17.js";
11
11
  import { E as pairedBootstrap } from "./statistics-ByxzSiOM.js";
12
- import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-D4I90_vR.js";
13
- import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-CYtcIF2V.js";
12
+ import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-CKtTpRhv.js";
13
+ import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-B181aMF9.js";
14
14
  import { z } from "zod";
15
15
  import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
16
16
  import * as nodePath from "node:path";
17
17
  import { dirname, isAbsolute, relative, resolve, sep } from "node:path";
18
18
  import { createHash, randomUUID } from "node:crypto";
19
+ import { request } from "node:http";
19
20
  import { link, lstat, mkdir, open, readFile, readdir, realpath, stat, unlink } from "node:fs/promises";
20
21
  import { arch, platform } from "node:os";
21
22
  import { TextDecoder as TextDecoder$1 } from "node:util";
22
23
  import { pathToFileURL } from "node:url";
24
+ import { request as request$1 } from "node:https";
23
25
  //#region src/analyst/benchmark-dataset-utils.ts
24
26
  function normalizeBenchmarkLabel(value) {
25
27
  const normalized = nonEmpty$1(value, "benchmark label").toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g, "");
@@ -706,7 +708,12 @@ const usageSchema = z.strictObject({
706
708
  calls: nonNegativeInteger.nullable(),
707
709
  tokens: tokenUsageSchema.nullable(),
708
710
  cost: costSchema,
709
- knownCostUsd: nonNegativeNumber.optional()
711
+ knownCostUsd: nonNegativeNumber.optional(),
712
+ partialTokens: z.strictObject({
713
+ input: nonNegativeInteger.nullable(),
714
+ output: nonNegativeInteger.nullable()
715
+ }).optional(),
716
+ tokensEstimated: z.boolean().optional()
710
717
  });
711
718
  const findingScoreSchema = z.strictObject({
712
719
  expectedIssueCount: nonNegativeInteger,
@@ -1200,7 +1207,7 @@ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
1200
1207
  "package.json",
1201
1208
  "pnpm-lock.yaml"
1202
1209
  ]);
1203
- const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "f0d788f72ea83b6bd485f2ae61e58dda7e1d87ec79284ce245b3a7f0cc7ca812";
1210
+ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "4785a1e0d785431f287362808dae777cf73125c551c18260117a711a5f2a6263";
1204
1211
  /** The published benchmark evidence was produced at this package version, by
1205
1212
  * the retired one-shot direct runner, before trace analysts moved to the
1206
1213
  * recursive DSPy RLM engine. Both evidence digests below are historical facts
@@ -1239,6 +1246,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1239
1246
  "src/analyst/benchmark-real-model.ts",
1240
1247
  "src/analyst/benchmark-report.ts",
1241
1248
  "src/analyst/benchmark-response-cache.ts",
1249
+ "src/analyst/benchmark-runner-prime.ts",
1242
1250
  "src/analyst/benchmark-scoring.ts",
1243
1251
  "src/analyst/benchmark-summary.ts",
1244
1252
  "src/analyst/benchmark-verification-artifacts.ts",
@@ -1251,6 +1259,8 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1251
1259
  "src/analyst/finding-subject.ts",
1252
1260
  "src/analyst/kind-factory.ts",
1253
1261
  "src/analyst/parse-tolerant.ts",
1262
+ "src/analyst/prime-bridge-transport.ts",
1263
+ "src/analyst/prime-protocol.ts",
1254
1264
  "src/analyst/tool-groups.ts",
1255
1265
  "src/analyst/trace-tool-callback.ts",
1256
1266
  "src/analyst/types.ts",
@@ -1299,7 +1309,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1299
1309
  "src/trace/raw-provider-sink.ts",
1300
1310
  "src/verdict-cache.ts"
1301
1311
  ]);
1302
- const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "67cd83aaaa158d44a9fa32e172cbe68fa97f01ae673ccb117469688b628bb9b1";
1312
+ const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "46af7a98d6df73d845394827283390bac7a2fb9712f418b14b1093f19a21faea";
1303
1313
  function analystBenchmarkImplementationDigest() {
1304
1314
  return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256;
1305
1315
  }
@@ -2437,7 +2447,7 @@ function createLocalRunReceipt(config, paths) {
2437
2447
  traceDir: resolve(config.traceDir),
2438
2448
  ...config.artifactDir ? { artifactDir: resolve(config.artifactDir) } : {},
2439
2449
  outputDir: paths.directory,
2440
- modelOwnerModule: config.modelOwnerModule
2450
+ ...config.modelOwnerModule === void 0 ? {} : { modelOwnerModule: config.modelOwnerModule }
2441
2451
  },
2442
2452
  command: config.command,
2443
2453
  environment: {
@@ -3596,7 +3606,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
3596
3606
  const maxModelRequestBytes = config.maxModelRequestBytes ?? 16 * 1024 * 1024;
3597
3607
  const maxModelResponseBytes = config.maxModelResponseBytes ?? 4 * 1024 * 1024;
3598
3608
  const modelRequestTimeoutMs = config.modelRequestTimeoutMs ?? timeoutMs;
3599
- const pricing = config.pricing ?? pricingForModel$1(model);
3609
+ const pricing = config.pricing ?? pricingForModel$2(model);
3600
3610
  const maxCostUsd = config.maxCostUsdPerAnalysis ?? 1;
3601
3611
  const costLedger = config.costLedger ?? new CostLedger();
3602
3612
  const durability = config.durability ? {
@@ -3608,7 +3618,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
3608
3618
  return {
3609
3619
  id: "direct",
3610
3620
  async analyze(input, context) {
3611
- const trajectoryId = trajectoryIdFromCaseId$1(dataset, context.caseId);
3621
+ const trajectoryId = trajectoryIdFromCaseId$2(dataset, context.caseId);
3612
3622
  const costTags = {
3613
3623
  analystId: actor,
3614
3624
  benchmarkCaseId: context.caseId,
@@ -3920,7 +3930,7 @@ function costReceiptMetadata(receipt) {
3920
3930
  estimatedCostUsd: null
3921
3931
  };
3922
3932
  }
3923
- function pricingForModel$1(model) {
3933
+ function pricingForModel$2(model) {
3924
3934
  const pricing = resolveModelPricing(model);
3925
3935
  if (!pricing) throw new Error(`no pricing is configured for '${model}'; provide PublicAnalystBenchmarkModelConfig.pricing`);
3926
3936
  return {
@@ -4096,7 +4106,7 @@ async function prepareSingleTraceContext(store, context) {
4096
4106
  });
4097
4107
  }
4098
4108
  }
4099
- function trajectoryIdFromCaseId$1(dataset, caseId) {
4109
+ function trajectoryIdFromCaseId$2(dataset, caseId) {
4100
4110
  const prefix = dataset === "agentrx" ? "agentrx:" : "codetrace:";
4101
4111
  if (!caseId.startsWith(prefix) || caseId.length === prefix.length) throw new Error(`unexpected ${dataset} benchmark case id '${caseId}'`);
4102
4112
  return caseId.slice(prefix.length);
@@ -4245,7 +4255,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4245
4255
  maxToolCalls: config.dspyRlm?.maxToolCalls ?? 80,
4246
4256
  maxOutputChars: config.dspyRlm?.maxOutputChars ?? 8e3
4247
4257
  };
4248
- const pricing = config.pricing ?? pricingForModel(config.model);
4258
+ const pricing = config.pricing ?? pricingForModel$1(config.model);
4249
4259
  const engine = createDspyRlmTraceEngine({
4250
4260
  call: config.call,
4251
4261
  callRef: config.callRef,
@@ -4278,7 +4288,7 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
4278
4288
  return {
4279
4289
  id: "dspy-rlm",
4280
4290
  async analyze(input, context) {
4281
- const trajectoryId = trajectoryIdFromCaseId(dataset, context.caseId);
4291
+ const trajectoryId = trajectoryIdFromCaseId$1(dataset, context.caseId);
4282
4292
  const tags = {
4283
4293
  benchmarkCaseId: context.caseId,
4284
4294
  benchmarkRepetition: String(context.repetition)
@@ -4518,7 +4528,7 @@ function publicBenchmarkDefinition(dataset, limits, instructions) {
4518
4528
  limits
4519
4529
  };
4520
4530
  }
4521
- function pricingForModel(model) {
4531
+ function pricingForModel$1(model) {
4522
4532
  const pricing = resolveModelPricing(model);
4523
4533
  if (!pricing) throw new Error(`no pricing is configured for '${model}'; provide PublicAnalystBenchmarkModelConfig.pricing`);
4524
4534
  return {
@@ -4526,7 +4536,7 @@ function pricingForModel(model) {
4526
4536
  outputUsdPerMillion: pricing.output * 1e3
4527
4537
  };
4528
4538
  }
4529
- function trajectoryIdFromCaseId(dataset, caseId) {
4539
+ function trajectoryIdFromCaseId$1(dataset, caseId) {
4530
4540
  const prefix = dataset === "agentrx" ? "agentrx:" : "codetrace:";
4531
4541
  if (!caseId.startsWith(prefix) || caseId.length === prefix.length) throw new Error(`unexpected ${dataset} benchmark case id '${caseId}'`);
4532
4542
  return caseId.slice(prefix.length);
@@ -4843,6 +4853,832 @@ function rootAgent(row) {
4843
4853
  return scalarDistributionValue(row.failures.find((failure) => String(failure.failure_id) === String(rootCauseId))?.failed_agent);
4844
4854
  }
4845
4855
  //#endregion
4856
+ //#region src/analyst/prime-bridge-transport.ts
4857
+ /**
4858
+ * Default transport on node:http/node:https rather than fetch: undici's fixed
4859
+ * response-header timeout kills prime calls that legitimately run past five
4860
+ * minutes, so the request's AbortSignal is the only deadline.
4861
+ */
4862
+ function nodeHttpPrimeBridgeTransport() {
4863
+ return ({ url, body, signal }) => {
4864
+ const target = new URL(url);
4865
+ if (target.protocol !== "http:" && target.protocol !== "https:") throw new TypeError(`bridge URL must be http: or https:, got ${target.protocol}`);
4866
+ const send = target.protocol === "https:" ? request$1 : request;
4867
+ const encoded = JSON.stringify(body);
4868
+ return new Promise((resolvePromise, rejectPromise) => {
4869
+ const req = send({
4870
+ hostname: target.hostname,
4871
+ port: target.port,
4872
+ path: `${target.pathname}${target.search}`,
4873
+ method: "POST",
4874
+ headers: {
4875
+ "content-type": "application/json",
4876
+ "content-length": Buffer.byteLength(encoded)
4877
+ },
4878
+ signal
4879
+ }, (res) => {
4880
+ const chunks = [];
4881
+ res.on("data", (chunk) => chunks.push(chunk));
4882
+ res.on("end", () => resolvePromise({
4883
+ status: res.statusCode ?? 0,
4884
+ text: Buffer.concat(chunks).toString("utf8")
4885
+ }));
4886
+ res.on("error", rejectPromise);
4887
+ });
4888
+ req.on("error", rejectPromise);
4889
+ req.end(encoded);
4890
+ });
4891
+ };
4892
+ }
4893
+ //#endregion
4894
+ //#region src/analyst/prime-protocol.ts
4895
+ function buildPrimePrompt(spec) {
4896
+ return [
4897
+ `QUESTION: ${spec.question}`,
4898
+ "",
4899
+ ...spec.taskDefinition === void 0 ? [] : [
4900
+ "TASK DEFINITION:",
4901
+ spec.taskDefinition,
4902
+ ""
4903
+ ],
4904
+ ...spec.contractLines,
4905
+ "",
4906
+ spec.trajectoryHeader,
4907
+ spec.renderedTrajectory,
4908
+ ...spec.trailer === void 0 ? [] : ["", spec.trailer]
4909
+ ].join("\n");
4910
+ }
4911
+ /** Carries the malformed reply and the contract — never the trajectory. */
4912
+ function buildPrimeRepairPrompt(spec) {
4913
+ return [
4914
+ "Your previous reply to a trace-analysis task was structurally malformed and could not be parsed",
4915
+ `(${spec.defect}). Below is your previous reply verbatim. Re-emit ONLY the corrected JSON — one`,
4916
+ "fenced ```json block, no other text, no tools. The JSON object has exactly two fields:",
4917
+ ...spec.repairContractLines,
4918
+ "",
4919
+ "PREVIOUS REPLY:",
4920
+ spec.previousReply
4921
+ ].join("\n");
4922
+ }
4923
+ /**
4924
+ * Recover the reply's JSON object.
4925
+ *
4926
+ * Distinct from `extractJsonPayload` in ../llm-client, which serves a response
4927
+ * that DECLARES a JSON root and therefore must not scan onward. A prime reply
4928
+ * is prose plus a fenced block, and when the model emits several fences the
4929
+ * last one is its answer — so fences are scanned in reverse, and only then is a
4930
+ * brace-to-brace slice tried.
4931
+ */
4932
+ function extractPrimeJsonObject(text) {
4933
+ const direct = parsePrimeJsonObject(text);
4934
+ if (direct) return direct;
4935
+ const fenced = [...text.matchAll(/```(?:json)?\s*\n?([\s\S]*?)```/g)];
4936
+ for (let index = fenced.length - 1; index >= 0; index -= 1) {
4937
+ const candidate = parsePrimeJsonObject(fenced[index][1]);
4938
+ if (candidate) return candidate;
4939
+ }
4940
+ const start = text.indexOf("{");
4941
+ const end = text.lastIndexOf("}");
4942
+ if (start >= 0 && end > start) {
4943
+ const candidate = parsePrimeJsonObject(text.slice(start, end + 1));
4944
+ if (candidate) return candidate;
4945
+ }
4946
+ return null;
4947
+ }
4948
+ /** Why the reply cannot be read as a prime answer, or null when it can. */
4949
+ function primeReplyDefect(parsed, rowsField) {
4950
+ if (parsed === null) return "no parseable JSON object";
4951
+ if (!Array.isArray(parsed[rowsField])) return `JSON has no "${rowsField}" array`;
4952
+ return null;
4953
+ }
4954
+ function parsePrimeJsonObject(text) {
4955
+ try {
4956
+ const value = JSON.parse(text.trim());
4957
+ return typeof value === "object" && value !== null && !Array.isArray(value) ? value : null;
4958
+ } catch {
4959
+ return null;
4960
+ }
4961
+ }
4962
+ function emptyPrimeRawUsage() {
4963
+ return {
4964
+ calls: null,
4965
+ inputTokens: null,
4966
+ outputTokens: null,
4967
+ bridgeEstimated: false
4968
+ };
4969
+ }
4970
+ /** Read the bridge's OpenAI-shaped `usage` object. */
4971
+ function normalizePrimeUsage(raw) {
4972
+ if (typeof raw !== "object" || raw === null) return emptyPrimeRawUsage();
4973
+ const record = raw;
4974
+ return {
4975
+ calls: tokenCountOrNull(record.model_requests),
4976
+ inputTokens: tokenCountOrNull(record.prompt_tokens),
4977
+ outputTokens: tokenCountOrNull(record.completion_tokens),
4978
+ bridgeEstimated: record.estimated === true
4979
+ };
4980
+ }
4981
+ /**
4982
+ * Sum two turns. Each side poisons independently: two turns that both report
4983
+ * input and neither report output yield a real input total beside a null
4984
+ * output, because discarding a measured count is as wrong as inventing one.
4985
+ */
4986
+ function mergePrimeRawUsage(a, b) {
4987
+ return {
4988
+ calls: sumOrNull(a.calls, b.calls),
4989
+ inputTokens: sumOrNull(a.inputTokens, b.inputTokens),
4990
+ outputTokens: sumOrNull(a.outputTokens, b.outputTokens),
4991
+ bridgeEstimated: a.bridgeEstimated || b.bridgeEstimated
4992
+ };
4993
+ }
4994
+ function sumOrNull(a, b) {
4995
+ return a !== null && b !== null ? a + b : null;
4996
+ }
4997
+ function tokenCountOrNull(value) {
4998
+ return typeof value === "number" && Number.isSafeInteger(value) && value >= 0 ? value : null;
4999
+ }
5000
+ /**
5001
+ * Bind raw prime usage to agent-eval's typed receipt.
5002
+ *
5003
+ * `RunTokenUsage.input` and `.output` are non-nullable, so a one-sided count
5004
+ * cannot round-trip through `tokens` without writing a zero nobody measured.
5005
+ * The complete-accounting field therefore stays null, the reported side is
5006
+ * carried verbatim in `partialTokens`, and its price becomes the receipt's
5007
+ * `knownCostUsd` lower bound.
5008
+ *
5009
+ * Only agent-eval calls this; consumers with no pricing table read
5010
+ * `PrimeRawUsage` directly.
5011
+ */
5012
+ function analystUsageReceiptFromPrimeUsage(usage, pricing) {
5013
+ const { calls, inputTokens, outputTokens, bridgeEstimated } = usage;
5014
+ const estimatedTokens = bridgeEstimated ? { tokensEstimated: true } : {};
5015
+ if (inputTokens !== null && outputTokens !== null) return {
5016
+ calls,
5017
+ tokens: {
5018
+ input: inputTokens,
5019
+ output: outputTokens
5020
+ },
5021
+ cost: {
5022
+ kind: "estimated",
5023
+ usd: priceTokens(inputTokens, outputTokens, pricing)
5024
+ },
5025
+ ...estimatedTokens
5026
+ };
5027
+ if (inputTokens === null && outputTokens === null) return {
5028
+ calls,
5029
+ tokens: null,
5030
+ cost: {
5031
+ kind: "uncaptured",
5032
+ usd: null
5033
+ },
5034
+ ...estimatedTokens
5035
+ };
5036
+ return {
5037
+ calls,
5038
+ tokens: null,
5039
+ partialTokens: {
5040
+ input: inputTokens,
5041
+ output: outputTokens
5042
+ },
5043
+ cost: {
5044
+ kind: "uncaptured",
5045
+ usd: null
5046
+ },
5047
+ knownCostUsd: priceTokens(inputTokens ?? 0, outputTokens ?? 0, pricing),
5048
+ ...estimatedTokens
5049
+ };
5050
+ }
5051
+ function priceTokens(input, output, pricing) {
5052
+ return (input * pricing.inputUsdPerMillion + output * pricing.outputUsdPerMillion) / 1e6;
5053
+ }
5054
+ /**
5055
+ * Run the protocol: one call, one bounded repair turn on a structurally
5056
+ * malformed reply, then decode. Zero valid rows from a well-formed reply is an
5057
+ * honest null, not a failure.
5058
+ */
5059
+ async function runPrimeExchange(options) {
5060
+ const { contract } = options;
5061
+ const turns = [];
5062
+ const repair = {
5063
+ attempted: false,
5064
+ succeeded: null
5065
+ };
5066
+ const first = await callPrimeTurn(options, options.prompt);
5067
+ if (!first.ok) return {
5068
+ ok: false,
5069
+ failure: first.failure,
5070
+ usage: mergeTurns(turns),
5071
+ turns,
5072
+ repair
5073
+ };
5074
+ turns.push({
5075
+ turn: "first",
5076
+ usage: first.usage,
5077
+ rawUsage: first.rawUsage
5078
+ });
5079
+ let reply = first.content;
5080
+ let parsed = extractPrimeJsonObject(reply);
5081
+ let defect = primeReplyDefect(parsed, contract.rowsField);
5082
+ if (defect !== null && options.repair) {
5083
+ repair.attempted = true;
5084
+ const second = await callPrimeTurn(options, buildPrimeRepairPrompt({
5085
+ defect,
5086
+ previousReply: reply,
5087
+ repairContractLines: contract.repairContractLines
5088
+ }));
5089
+ if (!second.ok) return {
5090
+ ok: false,
5091
+ failure: second.failure,
5092
+ usage: mergeTurns(turns),
5093
+ turns,
5094
+ repair,
5095
+ reply
5096
+ };
5097
+ turns.push({
5098
+ turn: "repair",
5099
+ usage: second.usage,
5100
+ rawUsage: second.rawUsage
5101
+ });
5102
+ reply = second.content;
5103
+ parsed = extractPrimeJsonObject(reply);
5104
+ defect = primeReplyDefect(parsed, contract.rowsField);
5105
+ repair.succeeded = defect === null;
5106
+ }
5107
+ const usage = mergeTurns(turns);
5108
+ if (defect !== null) return {
5109
+ ok: false,
5110
+ failure: {
5111
+ kind: "malformed-reply",
5112
+ message: `${defect} in prime reply${repair.attempted ? " even after the bounded repair turn" : ""}`
5113
+ },
5114
+ usage,
5115
+ turns,
5116
+ repair,
5117
+ reply
5118
+ };
5119
+ const rawRows = parsed[contract.rowsField];
5120
+ const rows = [];
5121
+ const rejected = [];
5122
+ let overflow = 0;
5123
+ rawRows.forEach((row, index) => {
5124
+ const decoded = contract.decodeRow(row, index);
5125
+ if (!decoded.ok) {
5126
+ rejected.push({
5127
+ index,
5128
+ reason: decoded.reason
5129
+ });
5130
+ return;
5131
+ }
5132
+ if (contract.maxRows !== void 0 && rows.length >= contract.maxRows) {
5133
+ overflow += 1;
5134
+ return;
5135
+ }
5136
+ rows.push(decoded.row);
5137
+ });
5138
+ const answer = parsed.answer;
5139
+ return {
5140
+ ok: true,
5141
+ answer: typeof answer === "string" ? answer : null,
5142
+ rows,
5143
+ rejected,
5144
+ reportedRows: rawRows.length,
5145
+ overflow,
5146
+ usage,
5147
+ turns,
5148
+ repair,
5149
+ reply
5150
+ };
5151
+ }
5152
+ async function callPrimeTurn(options, content) {
5153
+ const { transport, url, model, timeoutMs, signal } = options;
5154
+ const controller = new AbortController();
5155
+ const forwardAbort = () => controller.abort(signal?.reason);
5156
+ if (signal?.aborted) controller.abort(signal.reason);
5157
+ else signal?.addEventListener("abort", forwardAbort, { once: true });
5158
+ const deadline = setTimeout(() => controller.abort(), timeoutMs);
5159
+ let result;
5160
+ try {
5161
+ result = await transport({
5162
+ url,
5163
+ body: {
5164
+ model,
5165
+ messages: [{
5166
+ role: "user",
5167
+ content
5168
+ }]
5169
+ },
5170
+ signal: controller.signal
5171
+ });
5172
+ } catch (error) {
5173
+ if (signal?.aborted) return {
5174
+ ok: false,
5175
+ failure: {
5176
+ kind: "aborted",
5177
+ message: "prime exchange cancelled by the caller",
5178
+ cause: error
5179
+ }
5180
+ };
5181
+ if (controller.signal.aborted) return {
5182
+ ok: false,
5183
+ failure: {
5184
+ kind: "deadline",
5185
+ message: `bridge call exceeded ${timeoutMs}ms`
5186
+ }
5187
+ };
5188
+ return {
5189
+ ok: false,
5190
+ failure: {
5191
+ kind: "transport",
5192
+ message: `bridge transport failure: ${error instanceof Error ? error.message : String(error)}`
5193
+ }
5194
+ };
5195
+ } finally {
5196
+ clearTimeout(deadline);
5197
+ signal?.removeEventListener("abort", forwardAbort);
5198
+ }
5199
+ if (result.status !== 200) {
5200
+ const bodySnippet = result.text.slice(0, 500);
5201
+ return {
5202
+ ok: false,
5203
+ failure: {
5204
+ kind: "http-status",
5205
+ message: `bridge HTTP ${result.status}: ${bodySnippet}`,
5206
+ status: result.status,
5207
+ bodySnippet
5208
+ }
5209
+ };
5210
+ }
5211
+ let response;
5212
+ try {
5213
+ response = JSON.parse(result.text);
5214
+ } catch {
5215
+ return {
5216
+ ok: false,
5217
+ failure: {
5218
+ kind: "unparseable-json",
5219
+ message: `bridge returned unparseable JSON (${result.text.length} bytes)`
5220
+ }
5221
+ };
5222
+ }
5223
+ const replyContent = primeReplyContent(response);
5224
+ if (replyContent === null) return {
5225
+ ok: false,
5226
+ failure: {
5227
+ kind: "no-content",
5228
+ message: "bridge reply carries no message content"
5229
+ }
5230
+ };
5231
+ const rawUsage = primeReplyUsage(response);
5232
+ return {
5233
+ ok: true,
5234
+ content: replyContent,
5235
+ usage: normalizePrimeUsage(rawUsage),
5236
+ rawUsage
5237
+ };
5238
+ }
5239
+ /**
5240
+ * Fold from the FIRST turn, never from an empty receipt: an all-null identity
5241
+ * would poison every side it merged with and erase counts the bridge reported.
5242
+ */
5243
+ function mergeTurns(turns) {
5244
+ if (turns.length === 0) return emptyPrimeRawUsage();
5245
+ return turns.slice(1).reduce((total, turn) => mergePrimeRawUsage(total, turn.usage), turns[0].usage);
5246
+ }
5247
+ function primeReplyContent(response) {
5248
+ if (typeof response !== "object" || response === null) return null;
5249
+ const choices = response.choices;
5250
+ if (!Array.isArray(choices) || choices.length === 0) return null;
5251
+ const message = choices[0]?.message;
5252
+ if (typeof message !== "object" || message === null) return null;
5253
+ const content = message.content;
5254
+ return typeof content === "string" && content.length > 0 ? content : null;
5255
+ }
5256
+ function primeReplyUsage(response) {
5257
+ if (typeof response !== "object" || response === null) return null;
5258
+ return response.usage ?? null;
5259
+ }
5260
+ /**
5261
+ * Render, measure, fall back to the capped projection, re-measure, fail loud.
5262
+ *
5263
+ * Inline is the only delivery prime has, so an oversized trajectory is a
5264
+ * refusal rather than a silent truncation: dropping spans would understate the
5265
+ * trajectory and the analyst would answer a question about a different run.
5266
+ */
5267
+ async function projectPrimeTrajectory(source, limits) {
5268
+ let fetch = "full";
5269
+ let items = await source.full();
5270
+ if (items === null) {
5271
+ fetch = "capped";
5272
+ items = await source.capped();
5273
+ }
5274
+ let rendered = JSON.stringify(items);
5275
+ if (rendered.length > limits.maxInlineChars && fetch === "full") {
5276
+ fetch = "capped";
5277
+ items = await source.capped();
5278
+ rendered = JSON.stringify(items);
5279
+ }
5280
+ if (rendered.length > limits.maxInlineChars) return {
5281
+ ok: false,
5282
+ reason: `trajectory renders to ${rendered.length} chars even at ${source.cappedDescription}; inline delivery impossible`,
5283
+ renderedChars: rendered.length
5284
+ };
5285
+ return {
5286
+ ok: true,
5287
+ items,
5288
+ rendered,
5289
+ delivery: {
5290
+ mode: "inline-json",
5291
+ fetch,
5292
+ renderedChars: rendered.length
5293
+ }
5294
+ };
5295
+ }
5296
+ /**
5297
+ * Digest of everything a consumer can send to the bridge under the prime
5298
+ * protocol, recorded per observation so a prime result names the exact contract
5299
+ * that produced it.
5300
+ *
5301
+ * Computed over the ACTUALLY composed contract, so two consumers that both
5302
+ * stamp `analyst_id: 'prime'` while asking materially different questions get
5303
+ * different digests by construction. That is what makes 'prime' a reproducible
5304
+ * claim rather than a label.
5305
+ */
5306
+ function primeProtocolSha256(identity) {
5307
+ return createHash("sha256").update(JSON.stringify({
5308
+ kind: "prime-analyst-protocol",
5309
+ question: identity.question,
5310
+ taskPrompt: identity.taskDefinition ?? null,
5311
+ outputContract: identity.contractLines,
5312
+ repairContract: buildPrimeRepairPrompt({
5313
+ defect: "<defect>",
5314
+ previousReply: "<previous-reply>",
5315
+ repairContractLines: identity.repairContractLines
5316
+ }),
5317
+ limits: identity.limits
5318
+ })).digest("hex");
5319
+ }
5320
+ //#endregion
5321
+ //#region src/analyst/benchmark-runner-prime.ts
5322
+ /**
5323
+ * Prime analyst arm: the RLM coding agent reached through an OpenAI-compatible
5324
+ * cli-bridge solves the CodeTraceBench incorrect-step task as a one-shot trace
5325
+ * analyst.
5326
+ *
5327
+ * The runner consumes the same prepared benchmark cases every other runner
5328
+ * receives — the trace store already carries the appended final-verification
5329
+ * spans — and produces findings through the same published block expansion, so
5330
+ * a prime observation and a dspy-rlm observation differ only in which analyst
5331
+ * produced the blocks.
5332
+ *
5333
+ * The protocol itself — prompt composition, the bounded repair turn, reply
5334
+ * extraction, the projection ladder, usage normalization — lives in
5335
+ * `./prime-protocol`, which knows nothing about CodeTraceBench. This file is
5336
+ * the benchmark's binding to it: the block row grammar, the store-backed
5337
+ * projection source, and the benchmark observation shape.
5338
+ *
5339
+ * Trajectory delivery is inline JSON in the prompt. The dspy typed path binds
5340
+ * the viewTrace span projection as a REPL variable; prime has no REPL, so the
5341
+ * same projection is serialized into the prompt. When the full projection is
5342
+ * oversized the runner falls back to chunked viewSpans over the same
5343
+ * projection surface with a per-attribute byte cap, and fails loud if the
5344
+ * result still exceeds the inline budget.
5345
+ *
5346
+ * A structurally malformed reply gets ONE bounded repair turn (disable with
5347
+ * `repair: false`): a second stateless call carrying the malformed reply plus
5348
+ * the output contract — never the trajectory — mirroring the dspy arm's typed
5349
+ * repair so both arms face the same structured-output affordance. Still
5350
+ * malformed after repair = failed observation with a typed error, exactly how
5351
+ * a dspy-rlm failure is recorded. Zero valid blocks from a well-formed reply
5352
+ * is an honest null, not a failure.
5353
+ */
5354
+ const PRIME_ANALYST_ID = "prime";
5355
+ const PRIME_QUESTION = "Which assistant steps are incorrect under the CodeTraceBench definition?";
5356
+ /** Ceiling on the serialized trajectory JSON embedded in the prompt. */
5357
+ const MAX_INLINE_TRAJECTORY_CHARS = 36e4;
5358
+ /** Per-attribute projection cap used by the chunked viewSpans fallback. */
5359
+ const CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP = 1200;
5360
+ /** Minimal per-attribute cap used only to enumerate span ids in store order. */
5361
+ const SPAN_ID_ENUMERATION_ATTRIBUTE_BYTE_CAP = 64;
5362
+ /** viewSpans accepts at most 100 ids per call; 40 keeps each response bounded. */
5363
+ const VIEW_SPANS_CHUNK_SIZE = 40;
5364
+ const PRIME_SEVERITIES = /* @__PURE__ */ new Set([
5365
+ "critical",
5366
+ "high",
5367
+ "medium",
5368
+ "low",
5369
+ "info"
5370
+ ]);
5371
+ var PrimeBridgeTransportError = class extends Error {};
5372
+ var PrimeBridgeHttpError = class extends Error {
5373
+ status;
5374
+ constructor(status, bodySnippet) {
5375
+ super(`bridge HTTP ${status}: ${bodySnippet}`);
5376
+ this.status = status;
5377
+ }
5378
+ };
5379
+ var PrimeMalformedReplyError = class extends Error {};
5380
+ var PrimeTraceProjectionError = class extends Error {};
5381
+ /**
5382
+ * Short-strings rule: long reply strings get corrupted when the bridge splices
5383
+ * its backend's stream, so the contract forbids a rationale field and caps
5384
+ * every string the model must emit.
5385
+ */
5386
+ const PRIME_OUTPUT_CONTRACT_LINES = [
5387
+ "OUTPUT CONTRACT (supersedes any transport wording above — you have no trace tools and no REPL):",
5388
+ "You are a one-shot analyst. Every fact you need is in the TRAJECTORY JSON below.",
5389
+ "Do not run shell commands, do not read or write files, do not use any tools.",
5390
+ "Reply with EXACTLY one fenced ```json code block and no other fenced block. The JSON object has exactly two fields:",
5391
+ " \"answer\": string — ONE short sentence (max 300 chars) naming the latest failure evidence you traced from.",
5392
+ " \"blocks\": array (possibly empty) of failure blocks, each exactly:",
5393
+ " {\"first_step\": int, \"last_step\": int, \"consequence_step\": int,",
5394
+ " \"escape_status\": \"escaped\"|\"unescaped\",",
5395
+ " \"severity\": \"critical\"|\"high\"|\"medium\"|\"low\"|\"info\",",
5396
+ " \"claim\": string (ONE short sentence, max 200 chars),",
5397
+ " \"confidence\": number 0..1}",
5398
+ "Do NOT include a rationale field. Keep every string SHORT — long strings get corrupted in transport and void your work.",
5399
+ `Report at most 16 blocks; a block spans at most 12 steps.`,
5400
+ "Every step number must be the n of an existing assistant span with span_id \"step-<n>\" and kind \"LLM\" in the trajectory below; never cite TOOL, CHAIN, or AGENT spans.",
5401
+ "\"blocks\" is [] only for a clean trajectory."
5402
+ ];
5403
+ const PRIME_REPAIR_CONTRACT_LINES = [
5404
+ " \"answer\": string (ONE short sentence, max 300 chars)",
5405
+ " \"blocks\": array (possibly empty) of {\"first_step\": int, \"last_step\": int, \"consequence_step\": int,",
5406
+ " \"escape_status\": \"escaped\"|\"unescaped\", \"severity\": \"critical\"|\"high\"|\"medium\"|\"low\"|\"info\",",
5407
+ " \"claim\": string (max 200 chars), \"confidence\": number 0..1}",
5408
+ "No rationale field. Keep every string SHORT. Preserve the step numbers and verdicts of your previous reply exactly; shorten prose freely."
5409
+ ];
5410
+ const PRIME_PROTOCOL_IDENTITY = {
5411
+ question: PRIME_QUESTION,
5412
+ taskDefinition: CODE_TRACE_BENCH_ANALYST_PROMPT,
5413
+ contractLines: PRIME_OUTPUT_CONTRACT_LINES,
5414
+ repairContractLines: PRIME_REPAIR_CONTRACT_LINES,
5415
+ limits: {
5416
+ maxBlocks: 16,
5417
+ maxBlockSteps: 12,
5418
+ maxInlineTrajectoryChars: MAX_INLINE_TRAJECTORY_CHARS,
5419
+ chunkedProjectionAttributeByteCap: CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP
5420
+ }
5421
+ };
5422
+ /**
5423
+ * The block row grammar. No `maxRows`: the count cap belongs to
5424
+ * `expandCodeTraceFailureBlocks`, which drops the offending block and names it
5425
+ * in `diagnostics.droppedBlocks`, so capping here would erase that record.
5426
+ */
5427
+ const PRIME_BLOCK_CONTRACT = {
5428
+ rowsField: "blocks",
5429
+ contractLines: PRIME_OUTPUT_CONTRACT_LINES,
5430
+ repairContractLines: PRIME_REPAIR_CONTRACT_LINES,
5431
+ decodeRow(row) {
5432
+ const reason = blockRowDefect(row);
5433
+ if (reason !== null) return {
5434
+ ok: false,
5435
+ reason
5436
+ };
5437
+ return {
5438
+ ok: true,
5439
+ row: blockFromRow(row)
5440
+ };
5441
+ }
5442
+ };
5443
+ /**
5444
+ * Digest of everything this runner can send to the bridge, recorded per
5445
+ * observation so a prime result names the exact contract that produced it.
5446
+ */
5447
+ function primeAnalystProtocolSha256() {
5448
+ return primeProtocolSha256(PRIME_PROTOCOL_IDENTITY);
5449
+ }
5450
+ /** CodeTraceBench-only: the prompt and output contract speak its block grammar. */
5451
+ function createPrimeBenchmarkRunner(options) {
5452
+ const baseUrl = requiredString(options.baseUrl, "baseUrl").replace(/\/+$/, "");
5453
+ const model = requiredString(options.model, "model");
5454
+ const timeoutMs = positiveSafeInteger(options.timeoutMs, "timeoutMs");
5455
+ const repair = options.repair;
5456
+ if (typeof repair !== "boolean") throw new TypeError("repair must be a boolean");
5457
+ const pricing = options.pricing ?? pricingForModel(model);
5458
+ const transport = options.transport ?? nodeHttpPrimeBridgeTransport();
5459
+ const url = `${baseUrl}/v1/chat/completions`;
5460
+ return {
5461
+ id: PRIME_ANALYST_ID,
5462
+ async analyze(input, context) {
5463
+ const trajectoryId = trajectoryIdFromCaseId(context.caseId);
5464
+ let usage;
5465
+ let metadata = {
5466
+ analysisMode: "prime-rlm",
5467
+ engine: "prime",
5468
+ bridgeUrl: baseUrl,
5469
+ model,
5470
+ protocolSha256: primeAnalystProtocolSha256()
5471
+ };
5472
+ try {
5473
+ const store = input.traceStore;
5474
+ if (!store) throw new Error("codetracebench prime runner requires a trace store");
5475
+ const projection = await projectPrimeTrajectory(codeTraceProjectionSource(store, trajectoryId, context.signal ? { signal: context.signal } : void 0), { maxInlineChars: MAX_INLINE_TRAJECTORY_CHARS });
5476
+ if (!projection.ok) throw new PrimeTraceProjectionError(projection.reason);
5477
+ const delivery = {
5478
+ mode: projection.delivery.mode,
5479
+ fetch: projection.delivery.fetch === "full" ? "view-trace" : "view-spans-chunked",
5480
+ perAttributeByteCap: projection.delivery.fetch === "full" ? null : CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP,
5481
+ renderedChars: projection.delivery.renderedChars
5482
+ };
5483
+ metadata = {
5484
+ ...metadata,
5485
+ delivery
5486
+ };
5487
+ const prompt = buildCodeTracePrompt(trajectoryId, projection.items, projection.rendered);
5488
+ metadata = {
5489
+ ...metadata,
5490
+ promptChars: prompt.length
5491
+ };
5492
+ const outcome = await runPrimeExchange({
5493
+ contract: PRIME_BLOCK_CONTRACT,
5494
+ prompt,
5495
+ transport,
5496
+ url,
5497
+ model,
5498
+ timeoutMs,
5499
+ repair,
5500
+ ...context.signal ? { signal: context.signal } : {}
5501
+ });
5502
+ if (!outcome.ok && outcome.failure.kind === "aborted") throw abortCause(outcome.failure);
5503
+ if (outcome.turns.length > 0) {
5504
+ usage = analystUsageReceiptFromPrimeUsage(outcome.usage, pricing);
5505
+ metadata = {
5506
+ ...metadata,
5507
+ bridgeUsage: bridgeUsageFromTurns(outcome.turns)
5508
+ };
5509
+ }
5510
+ metadata = {
5511
+ ...metadata,
5512
+ repair: outcome.repair
5513
+ };
5514
+ if (!outcome.ok) {
5515
+ if (outcome.reply !== void 0) metadata = {
5516
+ ...metadata,
5517
+ reply: outcome.reply.slice(0, 4e3)
5518
+ };
5519
+ throw primeFailureError(outcome.failure);
5520
+ }
5521
+ const expanded = await expandCodeTraceFailureBlocks({
5522
+ trajectoryId,
5523
+ blocks: outcome.rows,
5524
+ store,
5525
+ analystId: PRIME_ANALYST_ID,
5526
+ ...context.signal ? { signal: context.signal } : {}
5527
+ });
5528
+ return {
5529
+ findings: expanded.findings,
5530
+ usage,
5531
+ metadata: {
5532
+ ...metadata,
5533
+ answer: outcome.answer,
5534
+ reportedRows: outcome.reportedRows,
5535
+ rejectedRows: outcome.rejected,
5536
+ blockDiagnostics: expanded.diagnostics
5537
+ }
5538
+ };
5539
+ } catch (error) {
5540
+ if (context.signal?.aborted) throw error;
5541
+ return {
5542
+ findings: [],
5543
+ ...usage ? { usage } : {},
5544
+ error: publicBenchmarkError(error, []),
5545
+ metadata
5546
+ };
5547
+ }
5548
+ }
5549
+ };
5550
+ }
5551
+ /**
5552
+ * The trace store, seen through the protocol's two-move projection contract:
5553
+ * the full viewTrace projection, or the chunked viewSpans projection at a
5554
+ * per-attribute byte cap.
5555
+ */
5556
+ function codeTraceProjectionSource(store, trajectoryId, context) {
5557
+ return {
5558
+ async full() {
5559
+ return (await store.viewTrace({ trace_id: trajectoryId }, context)).spans ?? null;
5560
+ },
5561
+ capped: () => projectSpansChunked(store, trajectoryId, context),
5562
+ cappedDescription: `per-attribute cap ${CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP}`
5563
+ };
5564
+ }
5565
+ /**
5566
+ * Chunked viewSpans projection for traces whose full viewTrace response is
5567
+ * oversized. Span ids come from a minimal-cap viewTrace in store order; every
5568
+ * id must project or the case fails loud — a silently dropped span would
5569
+ * understate the trajectory.
5570
+ */
5571
+ async function projectSpansChunked(store, trajectoryId, context) {
5572
+ const enumeration = await store.viewTrace({
5573
+ trace_id: trajectoryId,
5574
+ per_attribute_byte_cap: SPAN_ID_ENUMERATION_ATTRIBUTE_BYTE_CAP
5575
+ }, context);
5576
+ if (!enumeration.spans) throw new PrimeTraceProjectionError(`trace '${trajectoryId}' is oversized even at per-attribute cap ${SPAN_ID_ENUMERATION_ATTRIBUTE_BYTE_CAP}; cannot enumerate span ids`);
5577
+ const ids = [];
5578
+ const seen = /* @__PURE__ */ new Set();
5579
+ for (const span of enumeration.spans) if (typeof span.span_id === "string" && span.span_id.length > 0 && !seen.has(span.span_id)) {
5580
+ seen.add(span.span_id);
5581
+ ids.push(span.span_id);
5582
+ }
5583
+ if (ids.length === 0) throw new PrimeTraceProjectionError(`no span ids parsed from trace '${trajectoryId}'`);
5584
+ const projected = [];
5585
+ for (let index = 0; index < ids.length; index += VIEW_SPANS_CHUNK_SIZE) {
5586
+ const chunk = ids.slice(index, index + VIEW_SPANS_CHUNK_SIZE);
5587
+ const result = await store.viewSpans({
5588
+ trace_id: trajectoryId,
5589
+ span_ids: chunk,
5590
+ per_attribute_byte_cap: CHUNKED_PROJECTION_ATTRIBUTE_BYTE_CAP
5591
+ }, context);
5592
+ if (result.missing_span_ids.length > 0 || result.omitted_span_ids.length > 0 || result.spans.length !== chunk.length) throw new PrimeTraceProjectionError(`viewSpans projected ${result.spans.length}/${chunk.length} spans for chunk at ${index} of '${trajectoryId}'`);
5593
+ projected.push(...result.spans);
5594
+ }
5595
+ return projected;
5596
+ }
5597
+ function buildCodeTracePrompt(trajectoryId, spans, renderedSpans) {
5598
+ const stepSpans = spans.filter((span) => /^step-\d+$/.test(String(span.span_id)));
5599
+ if (stepSpans.length === 0) throw new PrimeTraceProjectionError(`no step-<n> spans in trace '${trajectoryId}'`);
5600
+ const finalVerification = spans.filter(isFinalVerificationSpan);
5601
+ return buildPrimePrompt({
5602
+ question: PRIME_QUESTION,
5603
+ taskDefinition: CODE_TRACE_BENCH_ANALYST_PROMPT,
5604
+ contractLines: PRIME_OUTPUT_CONTRACT_LINES,
5605
+ trajectoryHeader: `TRAJECTORY (trace_id ${trajectoryId}; ${stepSpans.length} assistant step spans; full span projection as JSON):`,
5606
+ renderedTrajectory: renderedSpans,
5607
+ trailer: finalVerification.length > 0 ? `FINAL VERIFICATION SPANS:\n${JSON.stringify(finalVerification)}` : "FINAL VERIFICATION: unavailable for this trajectory — trace backward from the latest failure evidence inside the trajectory itself."
5608
+ });
5609
+ }
5610
+ /** Map the protocol's terminal reason onto this benchmark's typed error classes. */
5611
+ function primeFailureError(failure) {
5612
+ switch (failure.kind) {
5613
+ case "http-status": return new PrimeBridgeHttpError(failure.status, failure.bodySnippet);
5614
+ case "malformed-reply": return new PrimeMalformedReplyError(failure.message);
5615
+ default: return new PrimeBridgeTransportError(failure.message);
5616
+ }
5617
+ }
5618
+ /** A cancelled run is not a result: the caller's error propagates unchanged. */
5619
+ function abortCause(failure) {
5620
+ return failure.cause instanceof Error ? failure.cause : new Error(failure.message);
5621
+ }
5622
+ function bridgeUsageFromTurns(turns) {
5623
+ return {
5624
+ first: turns.find((turn) => turn.turn === "first")?.rawUsage ?? null,
5625
+ repair: turns.find((turn) => turn.turn === "repair")?.rawUsage ?? null
5626
+ };
5627
+ }
5628
+ function isFinalVerificationSpan(span) {
5629
+ if (span.span_id.startsWith("benchmark-verification")) return true;
5630
+ const role = span.attributes["benchmark.evidence.role"];
5631
+ return typeof role === "string" && role.startsWith("final-verification");
5632
+ }
5633
+ function trajectoryIdFromCaseId(caseId) {
5634
+ if (!caseId.startsWith("codetrace:") || caseId.length === 10) throw new Error(`unexpected codetracebench benchmark case id '${caseId}'`);
5635
+ return caseId.slice(10);
5636
+ }
5637
+ function blockRowDefect(row) {
5638
+ if (typeof row !== "object" || row === null || Array.isArray(row)) return "row is not an object";
5639
+ const record = row;
5640
+ for (const field of [
5641
+ "first_step",
5642
+ "last_step",
5643
+ "consequence_step"
5644
+ ]) {
5645
+ const value = record[field];
5646
+ if (!Number.isInteger(value) || value < 1) return `${field} must be a positive integer`;
5647
+ }
5648
+ const firstStep = record.first_step;
5649
+ const lastStep = record.last_step;
5650
+ const consequenceStep = record.consequence_step;
5651
+ if (lastStep < firstStep) return "last_step < first_step";
5652
+ if (consequenceStep < firstStep) return "consequence_step < first_step";
5653
+ if (lastStep - firstStep + 1 > 12) return `block spans ${lastStep - firstStep + 1} steps (cap 12)`;
5654
+ if (record.escape_status !== "escaped" && record.escape_status !== "unescaped") return "escape_status must be escaped|unescaped";
5655
+ if (typeof record.severity !== "string" || !PRIME_SEVERITIES.has(record.severity)) return "severity outside the analyst severity enum";
5656
+ if (typeof record.claim !== "string" || record.claim.trim().length === 0 || record.claim.length > 2e3) return "claim must be a 1-2000 char string";
5657
+ if (typeof record.confidence !== "number" || !Number.isFinite(record.confidence) || record.confidence < 0 || record.confidence > 1) return "confidence must be 0..1";
5658
+ return null;
5659
+ }
5660
+ function blockFromRow(row) {
5661
+ const rationale = typeof row.rationale === "string" && row.rationale.trim().length > 0 ? row.rationale.trim().slice(0, 4e3) : void 0;
5662
+ return {
5663
+ firstStep: row.first_step,
5664
+ lastStep: row.last_step,
5665
+ consequenceStep: row.consequence_step,
5666
+ escapeStatus: row.escape_status,
5667
+ severity: row.severity,
5668
+ claim: row.claim.trim(),
5669
+ confidence: row.confidence,
5670
+ ...rationale === void 0 ? {} : { rationale }
5671
+ };
5672
+ }
5673
+ function pricingForModel(model) {
5674
+ const pricing = resolveModelPricing(model);
5675
+ if (!pricing) throw new Error(`no pricing is configured for '${model}'; provide PrimeBenchmarkRunnerOptions.pricing`);
5676
+ return {
5677
+ inputUsdPerMillion: pricing.input * 1e3,
5678
+ outputUsdPerMillion: pricing.output * 1e3
5679
+ };
5680
+ }
5681
+ //#endregion
4846
5682
  //#region src/analyst/benchmark-report.ts
4847
5683
  function renderAnalystBenchmarkMarkdown(result, comparisons = []) {
4848
5684
  const { provenance } = result;
@@ -4973,7 +5809,20 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
4973
5809
  return benchmarkExitCode(artifact.result, config.analyst);
4974
5810
  }
4975
5811
  if (await regularFileExists(paths.report)) throw new Error(`benchmark report exists without a completed result; refusing ambiguous resume: ${paths.report}`);
4976
- const createAnalystRunner = dependencies.createAnalystRunner ?? ((dataset, model) => config.analyst === "direct" ? createPublicBenchmarkDirectRunner(dataset, model) : createPublicBenchmarkRlmRunner(dataset, model));
5812
+ const createAnalystRunner = dependencies.createAnalystRunner ?? ((dataset, model) => {
5813
+ if (config.analyst === "prime") {
5814
+ if (!config.prime) throw new Error("analyst 'prime' is missing its bridge configuration");
5815
+ return createPrimeBenchmarkRunner({
5816
+ baseUrl: config.prime.bridgeUrl,
5817
+ model: model.model,
5818
+ timeoutMs: model.timeoutMs,
5819
+ repair: config.prime.repair,
5820
+ ...model.pricing ? { pricing: model.pricing } : {}
5821
+ });
5822
+ }
5823
+ const ownerModel = requireModelOwnerSettings(model);
5824
+ return config.analyst === "direct" ? createPublicBenchmarkDirectRunner(dataset, ownerModel) : createPublicBenchmarkRlmRunner(dataset, ownerModel);
5825
+ });
4977
5826
  const runners = [emptyPublicBenchmarkRunner(), createAnalystRunner(config.dataset, {
4978
5827
  ...config.model,
4979
5828
  costLedger,
@@ -5143,21 +5992,31 @@ Run the recursive DSPy trace analyst against public AgentRx or CodeTraceBench la
5143
5992
 
5144
5993
  Required:
5145
5994
  --dataset agentrx|codetracebench
5146
- --analyst dspy-rlm|direct Scored analyst. Default: dspy-rlm.
5995
+ --analyst dspy-rlm|direct|prime Scored analyst. Default: dspy-rlm.
5147
5996
  'direct' is the one-shot comparison arm.
5997
+ 'prime' is the RLM coding agent behind an
5998
+ OpenAI-compatible cli-bridge (codetracebench
5999
+ only; see docs/prime-analyst.md)
5148
6000
  --labels <dataset.json|dataset.jsonl>
5149
6001
  --trace-dir <one-trace-per-file OTLP JSONL directory>
5150
6002
  --artifact-dir <extracted artifact root> Required for CodeTraceBench
5151
6003
  --out <new output directory>
5152
6004
  --revision <full 40- or 64-character hex digest>
5153
6005
  --split <dataset split>
5154
- --model-owner-module <module> Module exporting createModelExecutionOwner;
5155
- the owner keeps provider credentials and policy
5156
- --model <provider model id>
6006
+ --model-owner-module <module> dspy-rlm|direct only. Module exporting
6007
+ createModelExecutionOwner; the owner keeps
6008
+ provider credentials and policy
6009
+ --model <provider model id> For prime, the bridge model id in
6010
+ <backend>/<provider>/<model> form, e.g.
6011
+ prime/zai/glm-5.2
5157
6012
  --limit <positive case count>
5158
6013
 
5159
6014
  Controls:
5160
6015
  --resume Continue an interrupted run in --out
6016
+ --bridge-url <url> prime only. OpenAI-compatible cli-bridge
6017
+ base URL. Default: http://localhost:4181
6018
+ --no-repair prime only. Disable the bounded repair turn
6019
+ for a structurally malformed reply
5161
6020
  --seed <integer> Case-selection and comparison seed. Default: 0
5162
6021
  --concurrency <positive integer> Parallel benchmark jobs. Default: 1
5163
6022
  --repetitions <positive integer> Runs per case and runner. Default: 1
@@ -5207,7 +6066,17 @@ async function parseCommandConfig(argv, env, dependencies) {
5207
6066
  const maxCostUsd = positiveFiniteFlag(flags, "max-cost-usd", 5);
5208
6067
  const python = flags.get("python")?.trim();
5209
6068
  const analyst = flags.get("analyst")?.trim() ?? "dspy-rlm";
5210
- if (analyst !== "dspy-rlm" && analyst !== "direct") throw new Error("--analyst must be 'dspy-rlm' or 'direct'");
6069
+ if (analyst !== "dspy-rlm" && analyst !== "direct" && analyst !== "prime") throw new Error("--analyst must be 'dspy-rlm', 'direct', or 'prime'");
6070
+ const bridgeUrl = flags.get("bridge-url")?.trim();
6071
+ if (bridgeUrl !== void 0 && analyst !== "prime") throw new Error("--bridge-url requires --analyst prime");
6072
+ if (bridgeUrl === "") throw new Error("--bridge-url must not be blank");
6073
+ if (flags.has("no-repair") && analyst !== "prime") throw new Error("--no-repair requires --analyst prime");
6074
+ if (analyst === "prime" && dataset !== "codetracebench") throw new Error("--analyst prime requires --dataset codetracebench; the prime runner speaks the CodeTraceBench failure-block contract");
6075
+ if (analyst === "prime" && flags.has("model-owner-module")) throw new Error("--model-owner-module is not used by --analyst prime; the cli-bridge owns model execution");
6076
+ const prime = analyst === "prime" ? {
6077
+ bridgeUrl: bridgeUrl ?? "http://localhost:4181",
6078
+ repair: !flags.has("no-repair")
6079
+ } : void 0;
5211
6080
  const rlmSamples = positiveFlag(flags, "rlm-samples", 1);
5212
6081
  if (rlmSamples > 1 && analyst !== "dspy-rlm") throw new Error("--rlm-samples above 1 requires --analyst dspy-rlm");
5213
6082
  const instructionsFile = flags.get("instructions-file")?.trim();
@@ -5215,13 +6084,13 @@ async function parseCommandConfig(argv, env, dependencies) {
5215
6084
  const instructionsOverride = instructionsFile ? readAnalystInstructionsOverride(instructionsFile) : void 0;
5216
6085
  if (rlmSamples > 1 && dataset !== "codetracebench") throw new Error("--rlm-samples above 1 requires --dataset codetracebench; step-level consensus is defined on its block grammar");
5217
6086
  const model = requiredFlag(flags, "model");
5218
- const modelOwnerModule = requiredFlag(flags, "model-owner-module");
5219
- const owner = await (dependencies.loadModelExecutionOwner ?? loadModelExecutionOwner)(modelOwnerModule, {
6087
+ const modelOwnerModule = analyst === "prime" ? void 0 : requiredFlag(flags, "model-owner-module");
6088
+ const owner = modelOwnerModule ? await (dependencies.loadModelExecutionOwner ?? loadModelExecutionOwner)(modelOwnerModule, {
5220
6089
  model,
5221
6090
  environment: Object.freeze({ ...env })
5222
- });
5223
- assertModelExecutionOwner(owner);
5224
- const pricing = owner.pricing ?? benchmarkModelPricing(model);
6091
+ }) : void 0;
6092
+ if (owner) assertModelExecutionOwner(owner);
6093
+ const pricing = owner?.pricing ?? benchmarkModelPricing(model);
5225
6094
  const maxOutputTokens = positiveFlag(flags, "max-output-tokens", 16384);
5226
6095
  const timeoutMs = positiveFlag(flags, "timeout-ms", 3e5);
5227
6096
  return {
@@ -5234,9 +6103,11 @@ async function parseCommandConfig(argv, env, dependencies) {
5234
6103
  revision: immutableRevision(requiredFlag(flags, "revision")),
5235
6104
  split: requiredFlag(flags, "split"),
5236
6105
  model: {
5237
- call: owner.call,
5238
- callRef: owner.callRef,
5239
- recordExecution: owner.recordExecution,
6106
+ ...owner ? {
6107
+ call: owner.call,
6108
+ callRef: owner.callRef,
6109
+ recordExecution: owner.recordExecution
6110
+ } : { callRef: `cli-bridge:${prime.bridgeUrl}` },
5240
6111
  model,
5241
6112
  maxOutputTokens,
5242
6113
  timeoutMs,
@@ -5274,11 +6145,22 @@ async function parseCommandConfig(argv, env, dependencies) {
5274
6145
  rlmSamples,
5275
6146
  maxCostUsd,
5276
6147
  maxArtifactBytes: positiveFlag(flags, "max-artifact-bytes", DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES),
5277
- modelOwnerModule,
6148
+ ...modelOwnerModule === void 0 ? {} : { modelOwnerModule },
6149
+ ...prime === void 0 ? {} : { prime },
5278
6150
  command: `agent-eval analyst-benchmark ${argv.filter((argument) => argument !== "--resume").map(shellQuote).join(" ")}`,
5279
6151
  resume: flags.has("resume")
5280
6152
  };
5281
6153
  }
6154
+ /** Fail-loud narrowing: the dspy-rlm and direct analysts require an owner call path. */
6155
+ function requireModelOwnerSettings(model) {
6156
+ const { call, recordExecution } = model;
6157
+ if (typeof call !== "function" || typeof recordExecution !== "function") throw new Error("model-owner execution is required for the dspy-rlm and direct analysts");
6158
+ return {
6159
+ ...model,
6160
+ call,
6161
+ recordExecution
6162
+ };
6163
+ }
5282
6164
  function parseFlags(argv) {
5283
6165
  const flags = /* @__PURE__ */ new Map();
5284
6166
  for (let index = 0; index < argv.length; index += 1) {
@@ -5304,6 +6186,8 @@ const KNOWN_FLAGS = /* @__PURE__ */ new Set([
5304
6186
  "resume",
5305
6187
  "dataset",
5306
6188
  "analyst",
6189
+ "bridge-url",
6190
+ "no-repair",
5307
6191
  "labels",
5308
6192
  "trace-dir",
5309
6193
  "artifact-dir",
@@ -5339,7 +6223,7 @@ const KNOWN_FLAGS = /* @__PURE__ */ new Set([
5339
6223
  "max-cost-usd",
5340
6224
  "max-artifact-bytes"
5341
6225
  ]);
5342
- const BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["resume"]);
6226
+ const BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["resume", "no-repair"]);
5343
6227
  function assertKnownFlags(flags) {
5344
6228
  for (const flag of flags.keys()) if (!KNOWN_FLAGS.has(flag)) throw new Error(`unknown analyst-benchmark flag: --${flag}`);
5345
6229
  }
@@ -5467,6 +6351,6 @@ function shellQuote(value) {
5467
6351
  return /^[a-zA-Z0-9_./:@%+=,-]+$/.test(value) ? value : `'${value.replaceAll("'", "'\\''")}'`;
5468
6352
  }
5469
6353
  //#endregion
5470
- export { normalizeAgentRxCategory as $, parseVerificationOutcome as A, analystBenchmarkDependencyLockDigest as B, MAX_INCORRECT_BLOCK_STEPS as C, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as D, publicBenchmarkSystemPrompt as E, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as F, ANALYST_BENCHMARK_OBSERVATIONS_FILE as G, ANALYST_BENCHMARK_COST_LEDGER_FILE as H, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as I, summarizeAgentRxCalibration as J, AGENT_RX_UPSTREAM_REVISION as K, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as L, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as M, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as N, appendVerificationArtifactsToOtlp as O, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as P, agentRxPredictionsToFindings as Q, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as R, MAX_INCORRECT_BLOCKS as S, publicBenchmarkRlmInstructions as T, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as U, analystBenchmarkImplementationDigest as V, ANALYST_BENCHMARK_MANIFEST_FILE as W, codeTracerPredictionsToFindings as X, codeTraceBenchCase as Y, agentRxBenchmarkCase as Z, compareAnalystRunners as _, preparePublicAnalystBenchmark as a, readAnalystInstructionsOverride as b, selectPublicBenchmarkRows as c, adaptPublicBenchmarkFindings as d, roundAgentRxStep as et, emptyPublicBenchmarkRunner as f, summarizeCodeTraceCalibration as g, renderCodeTraceCalibrationMarkdown as h, loadPublicBenchmarkRows as i, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as j, loadCodeTraceVerificationArtifacts as k, createPublicBenchmarkRlmRunner as l, readAnalystBenchmarkArtifact as m, runAnalystBenchmarkCommand as n, publicBenchmarkDistributions as o, expandCodeTraceFailureBlocks as p, renderAgentRxCalibrationMarkdown as q, renderAnalystBenchmarkMarkdown as r, publicBenchmarkSelectionReport as s, ANALYST_BENCHMARK_HELP as t, normalizeBenchmarkLabel as tt, createPublicBenchmarkDirectRunner as u, analystInstructionsOverrideFromText as v, publicBenchmarkProtocolSha256 as w, CODE_TRACE_BENCH_ANALYST_PROMPT as x, effectiveAnalystProtocolSha256 as y, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as z };
6354
+ export { ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as $, summarizeCodeTraceCalibration as A, publicBenchmarkSystemPrompt as B, createPublicBenchmarkRlmRunner as C, expandCodeTraceFailureBlocks as D, emptyPublicBenchmarkRunner as E, CODE_TRACE_BENCH_ANALYST_PROMPT as F, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as G, appendVerificationArtifactsToOtlp as H, MAX_INCORRECT_BLOCKS as I, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as J, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as K, MAX_INCORRECT_BLOCK_STEPS as L, analystInstructionsOverrideFromText as M, effectiveAnalystProtocolSha256 as N, readAnalystBenchmarkArtifact as O, readAnalystInstructionsOverride as P, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as Q, publicBenchmarkProtocolSha256 as R, selectPublicBenchmarkRows as S, adaptPublicBenchmarkFindings as T, loadCodeTraceVerificationArtifacts as U, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as V, parseVerificationOutcome as W, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as X, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as Y, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as Z, nodeHttpPrimeBridgeTransport as _, primeAnalystProtocolSha256 as a, ANALYST_BENCHMARK_OBSERVATIONS_FILE as at, publicBenchmarkDistributions as b, buildPrimeRepairPrompt as c, summarizeAgentRxCalibration as ct, mergePrimeRawUsage as d, agentRxBenchmarkCase as dt, analystBenchmarkDependencyLockDigest as et, normalizePrimeUsage as f, agentRxPredictionsToFindings as ft, runPrimeExchange as g, projectPrimeTrajectory as h, normalizeBenchmarkLabel as ht, createPrimeBenchmarkRunner as i, ANALYST_BENCHMARK_MANIFEST_FILE as it, compareAnalystRunners as j, renderCodeTraceCalibrationMarkdown as k, emptyPrimeRawUsage as l, codeTraceBenchCase as lt, primeReplyDefect as m, roundAgentRxStep as mt, runAnalystBenchmarkCommand as n, ANALYST_BENCHMARK_COST_LEDGER_FILE as nt, analystUsageReceiptFromPrimeUsage as o, AGENT_RX_UPSTREAM_REVISION as ot, primeProtocolSha256 as p, normalizeAgentRxCategory as pt, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as q, renderAnalystBenchmarkMarkdown as r, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as rt, buildPrimePrompt as s, renderAgentRxCalibrationMarkdown as st, ANALYST_BENCHMARK_HELP as t, analystBenchmarkImplementationDigest as tt, extractPrimeJsonObject as u, codeTracerPredictionsToFindings as ut, loadPublicBenchmarkRows as v, createPublicBenchmarkDirectRunner as w, publicBenchmarkSelectionReport as x, preparePublicAnalystBenchmark as y, publicBenchmarkRlmInstructions as z };
5471
6355
 
5472
- //# sourceMappingURL=benchmark-command-CQPKRUr-.js.map
6356
+ //# sourceMappingURL=benchmark-command-CA_NFOmy.js.map