@tangle-network/agent-eval 0.139.2 → 0.140.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/CHANGELOG.md +22 -0
  2. package/dist/analyst/index.d.ts +112 -18
  3. package/dist/analyst/index.d.ts.map +1 -1
  4. package/dist/analyst/index.js +7 -7
  5. package/dist/{benchmark-DDVdWcwA.d.ts → benchmark-DxaZfy0w.d.ts} +3 -3
  6. package/dist/{benchmark-DDVdWcwA.d.ts.map → benchmark-DxaZfy0w.d.ts.map} +1 -1
  7. package/dist/{benchmark-command-DzUZJl8M.js → benchmark-command-CK0UnXAD.js} +633 -339
  8. package/dist/benchmark-command-CK0UnXAD.js.map +1 -0
  9. package/dist/benchmarks/index.d.ts +1 -1
  10. package/dist/benchmarks/index.js +1 -1
  11. package/dist/{benchmarks-zxhy1QV3.js → benchmarks-HwoBE32G.js} +4 -4
  12. package/dist/{benchmarks-zxhy1QV3.js.map → benchmarks-HwoBE32G.js.map} +1 -1
  13. package/dist/campaign/index.d.ts +5 -5
  14. package/dist/campaign/index.js +3 -3
  15. package/dist/{campaign-DrS6_hLd.js → campaign-BzjYNYVZ.js} +5 -5
  16. package/dist/{campaign-DrS6_hLd.js.map → campaign-BzjYNYVZ.js.map} +1 -1
  17. package/dist/cli.js +2 -2
  18. package/dist/{client-BohnDFBq.d.ts → client-BoqGxEqx.d.ts} +4 -4
  19. package/dist/{client-BohnDFBq.d.ts.map → client-BoqGxEqx.d.ts.map} +1 -1
  20. package/dist/{completion-verifier-IPoP4fQO.d.ts → completion-verifier-D15NHYSk.d.ts} +5 -5
  21. package/dist/{completion-verifier-IPoP4fQO.d.ts.map → completion-verifier-D15NHYSk.d.ts.map} +1 -1
  22. package/dist/contract/index.d.ts +10 -10
  23. package/dist/contract/index.js +6 -6
  24. package/dist/control.d.ts +2 -2
  25. package/dist/{cost-ledger-CZ9diLxY.js → cost-ledger-DMFxsLKr.js} +22 -8
  26. package/dist/cost-ledger-DMFxsLKr.js.map +1 -0
  27. package/dist/{cost-ledger-DKgyIWRj.d.ts → cost-ledger-FuQvHxPm.d.ts} +8 -2
  28. package/dist/{cost-ledger-DKgyIWRj.d.ts.map → cost-ledger-FuQvHxPm.d.ts.map} +1 -1
  29. package/dist/{default-registry-B8vf7Rmf.d.ts → default-registry-Ci7wAAR8.d.ts} +5 -5
  30. package/dist/{default-registry-B8vf7Rmf.d.ts.map → default-registry-Ci7wAAR8.d.ts.map} +1 -1
  31. package/dist/{default-registry-BgJJItGr.js → default-registry-DCp-6hc-.js} +3 -3
  32. package/dist/{default-registry-BgJJItGr.js.map → default-registry-DCp-6hc-.js.map} +1 -1
  33. package/dist/{dspy-rlm-engine-DTkVyDX-.js → dspy-rlm-engine-Bkak4nzo.js} +11 -4
  34. package/dist/dspy-rlm-engine-Bkak4nzo.js.map +1 -0
  35. package/dist/{eval-campaign-BmptJj50.js → eval-campaign-YdkpWWoT.js} +2 -2
  36. package/dist/{eval-campaign-BmptJj50.js.map → eval-campaign-YdkpWWoT.js.map} +1 -1
  37. package/dist/{exact-types-MaaFcllV.d.ts → exact-types-B0lJV3tu.d.ts} +2 -2
  38. package/dist/{exact-types-MaaFcllV.d.ts.map → exact-types-B0lJV3tu.d.ts.map} +1 -1
  39. package/dist/{external-optimizer-contracts-BrxY2Sli.d.ts → external-optimizer-contracts-nb7c_WAR.d.ts} +12 -2
  40. package/dist/{external-optimizer-contracts-BrxY2Sli.d.ts.map → external-optimizer-contracts-nb7c_WAR.d.ts.map} +1 -1
  41. package/dist/{extract-usage-DZs601Va.js → extract-usage-C5vMw-0R.js} +2 -2
  42. package/dist/{extract-usage-DZs601Va.js.map → extract-usage-C5vMw-0R.js.map} +1 -1
  43. package/dist/{feedback-trajectory-BJUWOkJM.d.ts → feedback-trajectory-BCHqzLh3.d.ts} +3 -3
  44. package/dist/{feedback-trajectory-BJUWOkJM.d.ts.map → feedback-trajectory-BCHqzLh3.d.ts.map} +1 -1
  45. package/dist/fuzz.d.ts +1 -1
  46. package/dist/fuzz.js +1 -1
  47. package/dist/hosted/index.d.ts +3 -3
  48. package/dist/{index-BipJlj-C.d.ts → index-B798zyGh.d.ts} +5 -2
  49. package/dist/{index-BipJlj-C.d.ts.map → index-B798zyGh.d.ts.map} +1 -1
  50. package/dist/{index-_66rVpwN.d.ts → index-CKI1CXTL.d.ts} +5 -5
  51. package/dist/{index-_66rVpwN.d.ts.map → index-CKI1CXTL.d.ts.map} +1 -1
  52. package/dist/{index-CWOPCJiw.d.ts → index-D2enbqA0.d.ts} +2 -2
  53. package/dist/{index-CWOPCJiw.d.ts.map → index-D2enbqA0.d.ts.map} +1 -1
  54. package/dist/{index-BTm_P9aC.d.ts → index-DCP4I2Qx.d.ts} +10 -10
  55. package/dist/{index-BTm_P9aC.d.ts.map → index-DCP4I2Qx.d.ts.map} +1 -1
  56. package/dist/{index-CtR1xh4V.d.ts → index-DFLVtPZ9.d.ts} +3 -3
  57. package/dist/{index-CtR1xh4V.d.ts.map → index-DFLVtPZ9.d.ts.map} +1 -1
  58. package/dist/index.d.ts +24 -24
  59. package/dist/index.js +14 -14
  60. package/dist/{insight-report-Bu5Wi9tG.d.ts → insight-report-Bh_8ksel.d.ts} +4 -4
  61. package/dist/{insight-report-Bu5Wi9tG.d.ts.map → insight-report-Bh_8ksel.d.ts.map} +1 -1
  62. package/dist/{integrity-COTh3DTH.d.ts → integrity-DRXobPEs.d.ts} +2 -2
  63. package/dist/{integrity-COTh3DTH.d.ts.map → integrity-DRXobPEs.d.ts.map} +1 -1
  64. package/dist/{kind-factory-CFxA0JQX.js → kind-factory-DB7nIs35.js} +2 -2
  65. package/dist/{kind-factory-CFxA0JQX.js.map → kind-factory-DB7nIs35.js.map} +1 -1
  66. package/dist/{llm-client-bkztEfIx.js → llm-client-B3WXSH5Y.js} +2 -2
  67. package/dist/{llm-client-bkztEfIx.js.map → llm-client-B3WXSH5Y.js.map} +1 -1
  68. package/dist/meta-eval/index.d.ts +2 -2
  69. package/dist/multishot/index.d.ts +2 -2
  70. package/dist/openapi.json +1 -1
  71. package/dist/{release-report-fZarvIm-.d.ts → release-report-B_bQOHM-.d.ts} +3 -3
  72. package/dist/{release-report-fZarvIm-.d.ts.map → release-report-B_bQOHM-.d.ts.map} +1 -1
  73. package/dist/{replay-DjG4IG60.d.ts → replay-BqTgoioO.d.ts} +6 -6
  74. package/dist/{replay-DjG4IG60.d.ts.map → replay-BqTgoioO.d.ts.map} +1 -1
  75. package/dist/{replay-SA4OB7O7.js → replay-k2MsOmv5.js} +4 -4
  76. package/dist/{replay-SA4OB7O7.js.map → replay-k2MsOmv5.js.map} +1 -1
  77. package/dist/reporting.d.ts +4 -4
  78. package/dist/{researcher-BxhtGfKa.d.ts → researcher-C6lzl-rP.d.ts} +5 -5
  79. package/dist/{researcher-BxhtGfKa.d.ts.map → researcher-C6lzl-rP.d.ts.map} +1 -1
  80. package/dist/{reward-hacking-CqSLiV51.d.ts → reward-hacking-CEVVmy3h.d.ts} +2 -2
  81. package/dist/{reward-hacking-CqSLiV51.d.ts.map → reward-hacking-CEVVmy3h.d.ts.map} +1 -1
  82. package/dist/rl.d.ts +5 -5
  83. package/dist/rl.js +1 -1
  84. package/dist/rollout/index.d.ts +1 -1
  85. package/dist/{rubric-predictive-validity-DQBQj6uV.d.ts → rubric-predictive-validity-C2CthIfY.d.ts} +2 -2
  86. package/dist/{rubric-predictive-validity-DQBQj6uV.d.ts.map → rubric-predictive-validity-C2CthIfY.d.ts.map} +1 -1
  87. package/dist/{run-evidence-C4RcRQT5.d.ts → run-evidence-8Ou28QSa.d.ts} +3 -3
  88. package/dist/{run-evidence-C4RcRQT5.d.ts.map → run-evidence-8Ou28QSa.d.ts.map} +1 -1
  89. package/dist/{run-record-CztDMXVF.d.ts → run-record-Tb3TTtUn.d.ts} +2 -2
  90. package/dist/{run-record-CztDMXVF.d.ts.map → run-record-Tb3TTtUn.d.ts.map} +1 -1
  91. package/dist/{semantic-concept-judge-BuIJ9IfB.js → semantic-concept-judge-DJQtFr95.js} +3 -3
  92. package/dist/{semantic-concept-judge-BuIJ9IfB.js.map → semantic-concept-judge-DJQtFr95.js.map} +1 -1
  93. package/dist/{server-DaCpLfi0.js → server-Cu4M3NSO.js} +3 -3
  94. package/dist/{server-DaCpLfi0.js.map → server-Cu4M3NSO.js.map} +1 -1
  95. package/dist/{single-run-lock-BTTtPZ9N.js → single-run-lock-CiQThJxB.js} +22 -12
  96. package/dist/single-run-lock-CiQThJxB.js.map +1 -0
  97. package/dist/{skill-usage-B-BFS8M2.d.ts → skill-usage-CVVnoIx-.d.ts} +26 -10
  98. package/dist/skill-usage-CVVnoIx-.d.ts.map +1 -0
  99. package/dist/{skillopt-optimization-method-BbGnCC53.js → skillopt-optimization-method-CSBQ8Qma.js} +4 -4
  100. package/dist/{skillopt-optimization-method-BbGnCC53.js.map → skillopt-optimization-method-CSBQ8Qma.js.map} +1 -1
  101. package/dist/{skillopt-optimization-method-_s0Tub7Y.d.ts → skillopt-optimization-method-D1dqGzzH.d.ts} +11 -11
  102. package/dist/{skillopt-optimization-method-_s0Tub7Y.d.ts.map → skillopt-optimization-method-D1dqGzzH.d.ts.map} +1 -1
  103. package/dist/{statistics-B5d0Zd-z.d.ts → statistics-B4u_CiFd.d.ts} +2 -2
  104. package/dist/{statistics-B5d0Zd-z.d.ts.map → statistics-B4u_CiFd.d.ts.map} +1 -1
  105. package/dist/{store-otlp-DX4fGIcf.js → store-otlp-vRByAR6h.js} +2 -2
  106. package/dist/{store-otlp-DX4fGIcf.js.map → store-otlp-vRByAR6h.js.map} +1 -1
  107. package/dist/{summary-report-Cg7BifAM.d.ts → summary-report-o3eJ3gxG.d.ts} +3 -3
  108. package/dist/{summary-report-Cg7BifAM.d.ts.map → summary-report-o3eJ3gxG.d.ts.map} +1 -1
  109. package/dist/supervisor-run/index.d.ts +1 -1
  110. package/dist/supervisor-run/index.js +1 -1
  111. package/dist/{supervisor-run-B2EWUmQY.js → supervisor-run-D6A5oQw-.js} +18 -9
  112. package/dist/supervisor-run-D6A5oQw-.js.map +1 -0
  113. package/dist/{tool-groups-CdYq22lX.d.ts → tool-groups-DVQTy9lq.d.ts} +8 -8
  114. package/dist/{tool-groups-CdYq22lX.d.ts.map → tool-groups-DVQTy9lq.d.ts.map} +1 -1
  115. package/dist/traces.d.ts +6 -6
  116. package/dist/traces.js +4 -4
  117. package/dist/{types-uPrS6mD-.d.ts → types-BjMFz88h.d.ts} +2 -2
  118. package/dist/{types-uPrS6mD-.d.ts.map → types-BjMFz88h.d.ts.map} +1 -1
  119. package/dist/{types-DoEYskCd.d.ts → types-D3jh6F98.d.ts} +4 -4
  120. package/dist/{types-DoEYskCd.d.ts.map → types-D3jh6F98.d.ts.map} +1 -1
  121. package/dist/{types-BBFNHxSK.d.ts → types-Dk7PB7vh.d.ts} +5 -5
  122. package/dist/{types-BBFNHxSK.d.ts.map → types-Dk7PB7vh.d.ts.map} +1 -1
  123. package/dist/wire/index.d.ts +3 -3
  124. package/dist/wire/index.js +1 -1
  125. package/package.json +1 -1
  126. package/dist/benchmark-command-DzUZJl8M.js.map +0 -1
  127. package/dist/cost-ledger-CZ9diLxY.js.map +0 -1
  128. package/dist/dspy-rlm-engine-DTkVyDX-.js.map +0 -1
  129. package/dist/single-run-lock-BTTtPZ9N.js.map +0 -1
  130. package/dist/skill-usage-B-BFS8M2.d.ts.map +0 -1
  131. package/dist/supervisor-run-B2EWUmQY.js.map +0 -1
@@ -1,15 +1,15 @@
1
1
  import { c as ValidationError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
2
2
  import { s as resolveModelPricing } from "./metrics-C9YY1OcL.js";
3
- import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-CZ9diLxY.js";
4
- import { c as callLlmJson, f as maximumChargeForLlmRequest, l as costReceiptFromLlm, r as LlmResponseError, t as LlmCallError, u as costReceiptFromLlmError } from "./llm-client-bkztEfIx.js";
5
- import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-CFxA0JQX.js";
3
+ import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-DMFxsLKr.js";
4
+ import { c as callLlmJson, f as maximumChargeForLlmRequest, l as costReceiptFromLlm, r as LlmResponseError, t as LlmCallError, u as costReceiptFromLlmError } from "./llm-client-B3WXSH5Y.js";
5
+ import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-DB7nIs35.js";
6
6
  import { o as makeFinding, r as usageReceiptFromCostLedger } from "./usage-receipt-CgxMEBZq.js";
7
7
  import { i as hashCanonical, r as canonicalString } from "./canonical-D011XM8r.js";
8
- import { n as createRunCostLedger, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-BTTtPZ9N.js";
9
- import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-DTkVyDX-.js";
8
+ import { n as createRunCostLedger, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-CiQThJxB.js";
9
+ import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-Bkak4nzo.js";
10
10
  import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-Dxz0Rkwa.js";
11
11
  import { E as pairedBootstrap } from "./statistics-ByxzSiOM.js";
12
- import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-DX4fGIcf.js";
12
+ import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-vRByAR6h.js";
13
13
  import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-CYtcIF2V.js";
14
14
  import { z } from "zod";
15
15
  import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
@@ -1172,7 +1172,7 @@ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
1172
1172
  "package.json",
1173
1173
  "pnpm-lock.yaml"
1174
1174
  ]);
1175
- const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "1ca280998a9f3a416b40cadb9c14789ada490a727a4ec78491c92bb2022e6c95";
1175
+ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "1e4c763f2b791c57b37c0e326c56546b433a7a99c0bc7d064abdb58b9cbf0d81";
1176
1176
  /** The published benchmark evidence was produced at this package version, by
1177
1177
  * the retired one-shot direct runner, before trace analysts moved to the
1178
1178
  * recursive DSPy RLM engine. Both evidence digests below are historical facts
@@ -1203,6 +1203,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1203
1203
  "src/analyst/benchmark-public-data.ts",
1204
1204
  "src/analyst/benchmark-public-errors.ts",
1205
1205
  "src/analyst/benchmark-public-model.ts",
1206
+ "src/analyst/benchmark-public-prompt.ts",
1206
1207
  "src/analyst/benchmark-public-rlm.ts",
1207
1208
  "src/analyst/benchmark-public-types.ts",
1208
1209
  "src/analyst/benchmark-real-model.ts",
@@ -1265,7 +1266,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1265
1266
  "src/trace/otlp-attributes.ts",
1266
1267
  "src/trace/raw-provider-sink.ts"
1267
1268
  ]);
1268
- const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "5e47fc9d9f49c468d3ef06c5d552925e6a026f9e22a75056a9c8c5a2879744c0";
1269
+ const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "56dd4c7ed19fc5855f99ad238464ad95cb1bea155ad58c49dab3eb0f6cbe7d6a";
1269
1270
  function analystBenchmarkImplementationDigest() {
1270
1271
  return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256;
1271
1272
  }
@@ -1332,19 +1333,12 @@ async function validateCodeTraceFindingEvidence(options) {
1332
1333
  if (citations.length === 0) return;
1333
1334
  for (const citation of citations) if (!citation.location || citation.location.traceId !== options.trajectoryId) throw new Error(`model finding '${citation.findingId}' cites non-case evidence '${citation.evidence.uri}'`);
1334
1335
  const spanIds = [...new Set(citations.map((citation) => `step-${citation.location.step}`))];
1335
- const spans = /* @__PURE__ */ new Map();
1336
- for (let offset = 0; offset < spanIds.length; offset += TRACE_ANALYSIS_LIMITS.viewSpans) {
1337
- const requested = spanIds.slice(offset, offset + TRACE_ANALYSIS_LIMITS.viewSpans);
1338
- const result = await options.store.viewSpans({
1339
- trace_id: options.trajectoryId,
1340
- span_ids: requested
1341
- }, options.signal ? { signal: options.signal } : void 0);
1342
- if (result.missing_span_ids.length > 0 || result.omitted_span_ids.length > 0) {
1343
- const unavailable = [...result.missing_span_ids, ...result.omitted_span_ids];
1344
- throw new Error(`model finding evidence is unavailable in the case trace: ${unavailable.join(", ")}`);
1345
- }
1346
- for (const span of result.spans) spans.set(span.span_id, span);
1347
- }
1336
+ const { spans, missing } = await fetchTraceSpans(options.store, {
1337
+ trajectoryId: options.trajectoryId,
1338
+ spanIds,
1339
+ ...options.signal ? { signal: options.signal } : {}
1340
+ });
1341
+ if (missing.length > 0) throw new Error(`model finding evidence is unavailable in the case trace: ${missing.join(", ")}`);
1348
1342
  for (const citation of citations) {
1349
1343
  const spanId = `step-${citation.location.step}`;
1350
1344
  const span = spans.get(spanId);
@@ -1353,31 +1347,45 @@ async function validateCodeTraceFindingEvidence(options) {
1353
1347
  assertExactActionExcerpt(citation.findingId, citation.evidence, spanId, span.attributes.content);
1354
1348
  }
1355
1349
  }
1350
+ /**
1351
+ * Resolve assistant-step evidence for a trajectory.
1352
+ *
1353
+ * `steps` are claims the model made explicitly: an unresolvable one is a model
1354
+ * error and throws. `optionalSteps` are derived by the runner (a block's
1355
+ * interior, a block's consequence step), so an unresolvable one is simply
1356
+ * absent from the returned map and the caller decides what that means.
1357
+ */
1356
1358
  async function resolveAssistantStepEvidence(options) {
1357
- const steps = [...new Set(options.steps)];
1358
- for (const step of steps) if (!Number.isSafeInteger(step) || step < 1) throw new TypeError(`assistant evidence step must be a positive safe integer: ${step}`);
1359
- const spanIds = steps.map((step) => `step-${step}`);
1360
- const spans = /* @__PURE__ */ new Map();
1361
- for (let offset = 0; offset < spanIds.length; offset += TRACE_ANALYSIS_LIMITS.viewSpans) {
1362
- const requested = spanIds.slice(offset, offset + TRACE_ANALYSIS_LIMITS.viewSpans);
1363
- const result = await options.store.viewSpans({
1364
- trace_id: options.trajectoryId,
1365
- span_ids: requested
1366
- }, options.signal ? { signal: options.signal } : void 0);
1367
- if (result.missing_span_ids.length > 0 || result.omitted_span_ids.length > 0) {
1368
- const unavailable = [...result.missing_span_ids, ...result.omitted_span_ids];
1369
- throw new Error(`model selected unavailable assistant steps: ${unavailable.join(", ")}`);
1370
- }
1371
- for (const span of result.spans) spans.set(span.span_id, span);
1372
- }
1359
+ const required = [...new Set(options.steps)];
1360
+ const optional = [...new Set(options.optionalSteps ?? [])].filter((step) => !required.includes(step));
1361
+ for (const step of [...required, ...optional]) if (!Number.isSafeInteger(step) || step < 1) throw new TypeError(`assistant evidence step must be a positive safe integer: ${step}`);
1362
+ const steps = [...required, ...optional];
1363
+ if (steps.length === 0) return /* @__PURE__ */ new Map();
1364
+ const { spans, missing } = await fetchTraceSpans(options.store, {
1365
+ trajectoryId: options.trajectoryId,
1366
+ spanIds: steps.map((step) => `step-${step}`),
1367
+ ...options.signal ? { signal: options.signal } : {}
1368
+ });
1369
+ const missingRequired = missing.filter((spanId) => required.some((step) => `step-${step}` === spanId));
1370
+ if (missingRequired.length > 0) throw new Error(`model selected unavailable assistant steps: ${missingRequired.join(", ")}`);
1373
1371
  const evidence = /* @__PURE__ */ new Map();
1374
1372
  for (const step of steps) {
1375
1373
  const spanId = `step-${step}`;
1374
+ const optionalStep = optional.includes(step);
1376
1375
  const span = spans.get(spanId);
1377
- if (!span) throw new Error(`model selected missing assistant step '${spanId}'`);
1378
- if (span.kind !== "LLM") throw new Error(`model selected '${spanId}', which is ${span.kind}, not an assistant LLM span`);
1376
+ if (!span) {
1377
+ if (optionalStep) continue;
1378
+ throw new Error(`model selected missing assistant step '${spanId}'`);
1379
+ }
1380
+ if (span.kind !== "LLM") {
1381
+ if (optionalStep) continue;
1382
+ throw new Error(`model selected '${spanId}', which is ${span.kind}, not an assistant LLM span`);
1383
+ }
1379
1384
  const content = span.attributes.content;
1380
- if (typeof content !== "string" || content.trim().length === 0) throw new Error(`model selected '${spanId}' without action content`);
1385
+ if (typeof content !== "string" || content.trim().length === 0) {
1386
+ if (optionalStep) continue;
1387
+ throw new Error(`model selected '${spanId}' without action content`);
1388
+ }
1381
1389
  evidence.set(step, {
1382
1390
  kind: "span",
1383
1391
  uri: codeTraceStepEvidenceUri(options.trajectoryId, step),
@@ -1386,6 +1394,38 @@ async function resolveAssistantStepEvidence(options) {
1386
1394
  }
1387
1395
  return evidence;
1388
1396
  }
1397
+ /**
1398
+ * Read spans by id, paging over the store's byte-budget omissions.
1399
+ *
1400
+ * `omitted_span_ids` names spans that exist but did not fit the response
1401
+ * ceiling; the store guarantees at least one span lands per call, so
1402
+ * re-requesting exactly the omitted ids terminates. Only `missing_span_ids`
1403
+ * describes a span the trace does not contain.
1404
+ */
1405
+ async function fetchTraceSpans(store, options) {
1406
+ const spans = /* @__PURE__ */ new Map();
1407
+ const missing = [];
1408
+ const unique = [...new Set(options.spanIds)];
1409
+ const context = options.signal ? { signal: options.signal } : void 0;
1410
+ for (let offset = 0; offset < unique.length; offset += TRACE_ANALYSIS_LIMITS.viewSpans) {
1411
+ let pending = unique.slice(offset, offset + TRACE_ANALYSIS_LIMITS.viewSpans);
1412
+ while (pending.length > 0) {
1413
+ const result = await store.viewSpans({
1414
+ trace_id: options.trajectoryId,
1415
+ span_ids: pending
1416
+ }, context);
1417
+ for (const span of result.spans) spans.set(span.span_id, span);
1418
+ missing.push(...result.missing_span_ids);
1419
+ const omitted = result.omitted_span_ids.filter((spanId) => !spans.has(spanId));
1420
+ if (omitted.length >= pending.length) throw new Error(`trace '${options.trajectoryId}' cannot project spans within the store response budget: ${omitted.join(", ")}`);
1421
+ pending = omitted;
1422
+ }
1423
+ }
1424
+ return {
1425
+ spans,
1426
+ missing
1427
+ };
1428
+ }
1389
1429
  function scanValue(value, traceId, path, depth, serializedDepth) {
1390
1430
  if (depth > MAX_LABEL_SCAN_DEPTH) throw new Error(`trace '${traceId}' exceeds benchmark label scan depth at ${path}`);
1391
1431
  if (Array.isArray(value)) return 1 + value.reduce((count, entry, index) => count + scanValue(entry, traceId, `${path}[${index}]`, depth + 1, serializedDepth), 0);
@@ -1445,119 +1485,6 @@ function assertExactActionExcerpt(findingId, evidence, spanId, content) {
1445
1485
  if (!content.includes(excerpt)) throw new Error(`model finding '${findingId}' excerpt is not present in '${spanId}' action content`);
1446
1486
  }
1447
1487
  //#endregion
1448
- //#region src/analyst/benchmark-public-adapters.ts
1449
- function emptyPublicBenchmarkRunner() {
1450
- return {
1451
- id: "empty",
1452
- analyze() {
1453
- return {
1454
- findings: [],
1455
- usage: {
1456
- calls: 0,
1457
- tokens: {
1458
- input: 0,
1459
- output: 0
1460
- },
1461
- cost: {
1462
- kind: "observed",
1463
- usd: 0
1464
- }
1465
- },
1466
- metadata: { baseline: "emit-no-findings" }
1467
- };
1468
- }
1469
- };
1470
- }
1471
- function adaptPublicBenchmarkFindings(dataset, trajectoryId, findings, analystId) {
1472
- return dataset === "agentrx" ? adaptAgentRxFindings(trajectoryId, findings, analystId) : adaptCodeTraceFindings(trajectoryId, findings, analystId);
1473
- }
1474
- function adaptAgentRxFindings(trajectoryId, findings, analystId) {
1475
- if (findings.length === 0) return [];
1476
- if (findings.length !== 1) throw new Error(`AgentRx model analyst must emit zero or one root cause, received ${findings.length}`);
1477
- const source = findings[0];
1478
- if (!source.subject) throw new Error("AgentRx model analyst finding is missing its failure-category subject");
1479
- const steps = exactFindingSteps(trajectoryId, source);
1480
- if (steps.length !== 1) throw new Error(`AgentRx model analyst must cite exactly one root-cause step, received ${steps.length}`);
1481
- const [adapted] = agentRxPredictionsToFindings(trajectoryId, [{
1482
- failure_case: source.subject,
1483
- step_number: steps[0],
1484
- description: source.rationale ?? source.claim
1485
- }], {
1486
- analystId,
1487
- producedAt: source.produced_at,
1488
- confidence: source.confidence
1489
- });
1490
- if (!adapted) throw new Error("AgentRx output adapter produced no root-cause finding");
1491
- return [{
1492
- ...adapted,
1493
- metadata: {
1494
- ...adapted.metadata,
1495
- sourceFindingId: source.finding_id
1496
- }
1497
- }];
1498
- }
1499
- function adaptCodeTraceFindings(trajectoryId, findings, analystId) {
1500
- const clean = findings.filter((finding) => finding.subject === "clean");
1501
- if (clean.length > 0) {
1502
- if (findings.length !== 1) throw new Error("CodeTraceBench model analyst mixed a clean verdict with incorrect steps");
1503
- exactFindingSteps(trajectoryId, clean[0]);
1504
- return [];
1505
- }
1506
- const byStep = /* @__PURE__ */ new Map();
1507
- for (const source of findings) {
1508
- const steps = exactFindingSteps(trajectoryId, source);
1509
- for (const step of steps) {
1510
- if (byStep.has(step)) continue;
1511
- byStep.set(step, makeFinding({
1512
- analyst_id: analystId,
1513
- area: "incorrect",
1514
- subject: `incorrect-step-${step}`,
1515
- claim: `Step ${step} is incorrect. ${source.claim}`,
1516
- rationale: source.rationale,
1517
- severity: source.severity,
1518
- confidence: source.confidence,
1519
- evidence_refs: [{
1520
- kind: "span",
1521
- uri: codeTraceStepEvidenceUri(trajectoryId, step),
1522
- excerpt: source.evidence_refs.find((evidence) => evidence.uri === codeTraceStepEvidenceUri(trajectoryId, step))?.excerpt
1523
- }],
1524
- recommended_action: source.recommended_action,
1525
- metadata: { sourceFindingId: source.finding_id },
1526
- produced_at: source.produced_at,
1527
- id_basis: `incorrect-step-${step}`
1528
- }));
1529
- }
1530
- }
1531
- return [...byStep].sort(([left], [right]) => left - right).map(([, finding]) => finding);
1532
- }
1533
- function exactFindingSteps(trajectoryId, finding) {
1534
- if (finding.evidence_refs.length === 0) throw new Error(`model finding '${finding.finding_id}' has no step evidence`);
1535
- const steps = finding.evidence_refs.map((evidence) => {
1536
- const parsed = codeTraceStepFromEvidence(evidence.uri);
1537
- if (!parsed || parsed.traceId !== trajectoryId) throw new Error(`model finding '${finding.finding_id}' cites non-case evidence '${evidence.uri}'`);
1538
- return parsed.step;
1539
- });
1540
- return [...new Set(steps)];
1541
- }
1542
- //#endregion
1543
- //#region src/analyst/benchmark-public-types.ts
1544
- function requiredString(value, field) {
1545
- const trimmed = value.trim();
1546
- if (!trimmed) throw new TypeError(`${field} must be a non-empty string`);
1547
- return trimmed;
1548
- }
1549
- function positiveSafeInteger(value, field) {
1550
- if (!Number.isSafeInteger(value) || value < 1) throw new RangeError(`${field} must be a positive safe integer`);
1551
- return value;
1552
- }
1553
- function safeInteger(value, field) {
1554
- if (!Number.isSafeInteger(value)) throw new RangeError(`${field} must be a safe integer`);
1555
- return value;
1556
- }
1557
- function isRecord(value) {
1558
- return typeof value === "object" && value !== null && !Array.isArray(value);
1559
- }
1560
- //#endregion
1561
1488
  //#region src/analyst/benchmark-verification-outcome.ts
1562
1489
  const MAX_REPORTED_CHECKS = 20;
1563
1490
  const SWE_MULTI_NO_TEST_RESULTS = "After applying the fix patch, no test results were captured when executing the test command.";
@@ -2157,6 +2084,391 @@ function isNodeError$1(error, code) {
2157
2084
  return error instanceof Error && "code" in error && error.code === code;
2158
2085
  }
2159
2086
  //#endregion
2087
+ //#region src/analyst/benchmark-public-prompt.ts
2088
+ /** Widest contiguous failure block a model may report. The published corpus's
2089
+ * widest labeled block is 8 steps and its widest stage span is 9, so this bound
2090
+ * never binds honest enumeration; it caps how far one over-wide block can push
2091
+ * unlabeled steps into the precision denominator. */
2092
+ const MAX_INCORRECT_BLOCK_STEPS = 12;
2093
+ /** Most blocks a model may report for one trajectory. The published corpus's
2094
+ * densest case carries 4 disjoint labeled blocks. Together with the per-block
2095
+ * cap this bounds one case at 192 predicted steps without a second ceiling. */
2096
+ const MAX_INCORRECT_BLOCKS = 16;
2097
+ const TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS = [
2098
+ 4096,
2099
+ 2048,
2100
+ 1024,
2101
+ 512,
2102
+ 256,
2103
+ 128,
2104
+ 64
2105
+ ];
2106
+ const CODE_TRACE_BENCH_ANALYST_PROMPT = `Analyze exactly one coding-agent trajectory and its attached final verification.
2107
+ Your task is the CodeTraceBench incorrect-step task: identify every wrong state-changing action, bad hypothesis that drives an action, and regression.
2108
+ An incorrect step remains incorrect when the agent later recovers or the final verification passes; a trajectory that ends in success still contains every mistake the agent made along the way.
2109
+ Incorrect steps occur in contiguous failure blocks: one mistake plus every consecutive following step that commits to, compounds, or acts on it.
2110
+ Report each failure block as exactly one finding whose first_step is the block's first incorrect step and whose last_step is its last, covering every consecutive step between them.
2111
+ Set first_step to the first step that commits the mistake, not the step that planned it and not a later step that repeats it.
2112
+ Extend last_step one step at a time, and only while the next step independently satisfies the incorrect-step definition on its own action and its own following observation.
2113
+ Stop at the first step where the agent detects the problem, inspects it, or begins repairing it: a diagnostic probe, a test run that exposes the defect, or a repair action ends the block and is never inside it.
2114
+ A one-step block is a complete and correct answer.
2115
+ Every step inside a block is scored on its own: naming a correct step costs exactly as much as missing an incorrect one, and naming only the first step of a longer block forfeits every unnamed step.
2116
+ Report blocks separated by at least one correct step as separate findings, and never let two blocks overlap.
2117
+ Inspect the complete supplied trace data.
2118
+ Use the final-verification outcome as evidence about the final state, not as a rule for whether earlier steps were incorrect.
2119
+ For each candidate block, inspect every assistant action in it and its following observation.
2120
+ Admit a block only when you can point at the specific evidence it produced: name as consequence_step the step number whose action or observation shows the damage — a failing command, a wrong file state, a repeated failure, or rework the agent had to do because of this block. That step is the block's own last step when its observation already shows the damage, and a later step otherwise.
2121
+ When you cannot name that later step number from the trace you were given, drop the block; a plausible story about why a step looks wrong is not evidence that it was.
2122
+ Judge that consequence from the trajectory itself: a passing final verification is not evidence that a block caused nothing, and a failing final verification is not evidence that any particular block caused it.
2123
+ For every block, decide whether the agent escaped the failure.
2124
+ Mark escape_status "escaped" only when you can name the single later step that fully reversed the block, the agent needed no other step to recover, and nothing after that step revisits the same file, command, or hypothesis; write that step number in the rationale.
2125
+ Mark escape_status "unescaped" in every other case, including whenever you are unsure.
2126
+ A passing final verification never makes a block escaped: the agent may have made the mistake and repaired it over several steps, and those steps are still incorrect.
2127
+ Label a failed command when the assistant caused it through a wrong action or unsupported hypothesis.
2128
+ Label the later corrective action only when that action is itself wrong.
2129
+ Do not label a diagnostic probe merely because it exposes an earlier defect.
2130
+ Do not label a redundant but correct read or search; CodeTraceBench scores unuseful steps separately, and this run scores incorrect steps only.
2131
+ Do not label a step solely because final verification failed.
2132
+ When final verification is unavailable, use only directly observed trajectory evidence.
2133
+ Every step in a reported block MUST be the positive integer n from an existing assistant LLM span named step-<n>.
2134
+ Never select an EVALUATOR, TOOL, CHAIN, final-verification, benchmark-verification, or message-<n> span.
2135
+ Before emitting a finding, inspect every covered span's attributes.content and describe only the actions shown there.
2136
+ Report at most 16 blocks and at most 12 steps in one block; when more candidates than that exist, report the ones you can support with the clearest downstream evidence.
2137
+ When the trajectory has no incorrect steps, return an empty findings array.`;
2138
+ const AGENT_RX_PROMPT = `Analyze exactly one failed agent trajectory.
2139
+ Find the first unrecoverable critical failure, not every later symptom.
2140
+ Inspect the complete supplied trace data.
2141
+ Emit zero findings only when the trace does not contain enough evidence.
2142
+ Otherwise emit exactly one finding.
2143
+ Its category MUST be exactly one of:
2144
+ instruction-plan-adherence-failure
2145
+ invention-of-new-information
2146
+ invalid-invocation
2147
+ misinterpretation-of-tool-output-handoff-failure
2148
+ intent-plan-misalignment
2149
+ underspecified-user-intent
2150
+ intent-not-supported
2151
+ guardrails-triggered
2152
+ system-failure
2153
+ inconclusive
2154
+ Its step is the positive integer n from the first unrecoverable assistant span named step-<n>.`;
2155
+ const AGENT_RX_JSON_CONTRACT = `Each finding must contain only:
2156
+ - "step": a positive integer matching an existing assistant LLM span named step-<n>
2157
+ - "severity": "critical", "high", "medium", "low", or "info"
2158
+ - "claim": one sentence
2159
+ - "confidence": a number from 0 through 1
2160
+ - optional "rationale" and "recommended_action" strings
2161
+ - "category": one allowed failure category listed above`;
2162
+ const CODE_TRACE_JSON_CONTRACT = `Each finding is one contiguous failure block and must contain only:
2163
+ - "first_step": a positive integer, the block's first incorrect step, matching an existing assistant LLM span named step-<n>
2164
+ - "last_step": a positive integer >= first_step, the block's last incorrect step; every step from first_step through last_step must be an existing assistant LLM span, and a block spans at most 12 steps
2165
+ - "consequence_step": a positive integer >= first_step, the step whose action or following observation shows the damage this block caused; it may sit inside the block when the damage is already visible there
2166
+ - "escape_status": "escaped" only when one single later step fully reversed the block and nothing afterwards revisits it, "unescaped" otherwise and whenever you are unsure
2167
+ - "severity": "critical", "high", "medium", "low", or "info"
2168
+ - "claim": one sentence describing the block's failure
2169
+ - "confidence": a number from 0 through 1
2170
+ - optional "rationale" and "recommended_action" strings`;
2171
+ const AGENT_RX_RLM_CONTRACT = `Use the trace tools to inspect the action and its following observation.
2172
+ Emit exactly one finding whose subject is exactly one of the allowed failure categories.
2173
+ Cite exactly one assistant span named step-<n> as trace://<URL-encoded-trace-id>/span/step-<n>.
2174
+ The excerpt must quote the assistant action exactly.`;
2175
+ const CODE_TRACE_RLM_CONTRACT = `Use the trace tools rather than asking for the whole trajectory in the prompt.
2176
+ Keep retrieved trace objects in Python variables.
2177
+ Never print an entire trace, full source file, or more than 12000 characters in one iteration.
2178
+ Build a compact table of assistant step ids, actions, following observations, and final verification.
2179
+ Inspect suspicious steps with viewSpans or searchSpan instead of repeatedly printing the table.
2180
+ This runner emits no JSON fields, so the block is encoded in the finding's subject.
2181
+ Only findings_json is scored; your prose answer is ignored, so every incorrect block you identify must appear as a finding, never only in the answer.
2182
+ Emit exactly one finding per contiguous failure block.
2183
+ Set the finding's subject to incorrect-steps-<first_step>-<last_step>-<escape_status>-consequence-<consequence_step>, using the same four values the task defines; for a block covering only step 7 that the agent never escaped and whose damage shows at step 9, the subject is incorrect-steps-7-7-unescaped-consequence-9.
2184
+ The runner expands the block to one scored step per member and builds every scored citation itself.
2185
+ Cite the block's first step and its last step as trace://<URL-encoded-trace-id>/span/step-<n>, each excerpt an exact quote from that step's own action content.
2186
+ Give the rationale as the concrete downstream evidence visible at the consequence step.
2187
+ Submit as soon as every candidate failure block has a supported verdict.
2188
+ Return no finding for a clean trajectory.`;
2189
+ /** One-shot JSON transport prompt for the direct runner. */
2190
+ function publicBenchmarkSystemPrompt(dataset) {
2191
+ const fieldContract = dataset === "agentrx" ? AGENT_RX_JSON_CONTRACT : CODE_TRACE_JSON_CONTRACT;
2192
+ return `${publicBenchmarkTaskPrompt(dataset)}
2193
+
2194
+ ${fieldContract}
2195
+
2196
+ Return exactly one JSON object with:
2197
+ - "report": a concise evidence-based explanation, at most 4000 characters
2198
+ - "findings": the strict finding array
2199
+ Use an empty findings array when the trace does not support a finding.
2200
+ Do not return a bare array, markdown, trace URIs, copied excerpts, or fields not listed above.
2201
+ The runner constructs exact trace URIs and action previews from each selected step.`;
2202
+ }
2203
+ /** Tool-loop prompt for the recursive runner. Same task, subject-encoded block. */
2204
+ function publicBenchmarkRlmInstructions(dataset) {
2205
+ const outputContract = dataset === "agentrx" ? AGENT_RX_RLM_CONTRACT : CODE_TRACE_RLM_CONTRACT;
2206
+ return `${publicBenchmarkTaskPrompt(dataset)}
2207
+ ${outputContract}`;
2208
+ }
2209
+ function publicBenchmarkTaskPrompt(dataset) {
2210
+ return dataset === "agentrx" ? AGENT_RX_PROMPT : CODE_TRACE_BENCH_ANALYST_PROMPT;
2211
+ }
2212
+ /** Digest of every prompt a runner can send plus the shared transport limits.
2213
+ * Both runner contracts are hashed so an edit to either one changes the digest
2214
+ * a run records, whichever runner executed. */
2215
+ function publicBenchmarkProtocolSha256(dataset) {
2216
+ return sha256Digest(JSON.stringify({
2217
+ dataset,
2218
+ systemPrompt: publicBenchmarkSystemPrompt(dataset),
2219
+ rlmInstructions: publicBenchmarkRlmInstructions(dataset),
2220
+ transport: {
2221
+ attempts: 1,
2222
+ jsonMode: true,
2223
+ thinking: "disabled"
2224
+ },
2225
+ blockLimits: dataset === "agentrx" ? null : {
2226
+ maxBlocks: 16,
2227
+ maxBlockSteps: 12
2228
+ },
2229
+ traceProjectionAttributeByteCaps: TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS,
2230
+ evidence: {
2231
+ location: "model-selected-positive-integer-assistant-step",
2232
+ uri: "deterministic-trace-uri",
2233
+ excerpt: `exact-action-prefix-512`
2234
+ }
2235
+ }));
2236
+ }
2237
+ //#endregion
2238
+ //#region src/analyst/benchmark-public-adapters.ts
2239
+ function emptyPublicBenchmarkRunner() {
2240
+ return {
2241
+ id: "empty",
2242
+ analyze() {
2243
+ return {
2244
+ findings: [],
2245
+ usage: {
2246
+ calls: 0,
2247
+ tokens: {
2248
+ input: 0,
2249
+ output: 0
2250
+ },
2251
+ cost: {
2252
+ kind: "observed",
2253
+ usd: 0
2254
+ }
2255
+ },
2256
+ metadata: { baseline: "emit-no-findings" }
2257
+ };
2258
+ }
2259
+ };
2260
+ }
2261
+ async function adaptPublicBenchmarkFindings(options) {
2262
+ if (options.dataset === "agentrx") return {
2263
+ findings: adaptAgentRxFindings(options.trajectoryId, options.findings, options.analystId),
2264
+ diagnostics: void 0
2265
+ };
2266
+ return adaptCodeTraceFindings(options.trajectoryId, options.findings, options.analystId, options.store, options.signal);
2267
+ }
2268
+ function adaptAgentRxFindings(trajectoryId, findings, analystId) {
2269
+ if (findings.length === 0) return [];
2270
+ if (findings.length !== 1) throw new Error(`AgentRx model analyst must emit zero or one root cause, received ${findings.length}`);
2271
+ const source = findings[0];
2272
+ if (!source.subject) throw new Error("AgentRx model analyst finding is missing its failure-category subject");
2273
+ const steps = exactFindingSteps(trajectoryId, source);
2274
+ if (steps.length !== 1) throw new Error(`AgentRx model analyst must cite exactly one root-cause step, received ${steps.length}`);
2275
+ const [adapted] = agentRxPredictionsToFindings(trajectoryId, [{
2276
+ failure_case: source.subject,
2277
+ step_number: steps[0],
2278
+ description: source.rationale ?? source.claim
2279
+ }], {
2280
+ analystId,
2281
+ producedAt: source.produced_at,
2282
+ confidence: source.confidence
2283
+ });
2284
+ if (!adapted) throw new Error("AgentRx output adapter produced no root-cause finding");
2285
+ return [{
2286
+ ...adapted,
2287
+ metadata: {
2288
+ ...adapted.metadata,
2289
+ sourceFindingId: source.finding_id
2290
+ }
2291
+ }];
2292
+ }
2293
+ const CODE_TRACE_BLOCK_SUBJECT = /^incorrect-steps-(\d+)-(\d+)-(escaped|unescaped)-consequence-(\d+)$/;
2294
+ async function adaptCodeTraceFindings(trajectoryId, findings, analystId, store, signal) {
2295
+ const clean = findings.filter((finding) => finding.subject === "clean");
2296
+ if (clean.length > 0) {
2297
+ if (findings.length !== 1) throw new Error("CodeTraceBench model analyst mixed a clean verdict with incorrect steps");
2298
+ exactFindingSteps(trajectoryId, clean[0]);
2299
+ return {
2300
+ findings: [],
2301
+ diagnostics: emptyCodeTraceBlockDiagnostics()
2302
+ };
2303
+ }
2304
+ const blocks = [];
2305
+ const rejectedFindings = [];
2306
+ for (const source of findings) try {
2307
+ await validateCodeTraceFindingEvidence({
2308
+ trajectoryId,
2309
+ findings: [source],
2310
+ store,
2311
+ ...signal ? { signal } : {}
2312
+ });
2313
+ blocks.push(codeTraceBlockFromFinding(trajectoryId, source));
2314
+ } catch (error) {
2315
+ rejectedFindings.push(`${source.finding_id}: ${error instanceof Error ? error.message : String(error)}`);
2316
+ }
2317
+ const expanded = await expandCodeTraceFailureBlocks({
2318
+ trajectoryId,
2319
+ blocks,
2320
+ store,
2321
+ analystId,
2322
+ ...findings[0] ? { producedAt: findings[0].produced_at } : {},
2323
+ ...signal ? { signal } : {}
2324
+ });
2325
+ return {
2326
+ findings: expanded.findings,
2327
+ diagnostics: {
2328
+ ...expanded.diagnostics,
2329
+ rejectedFindings
2330
+ }
2331
+ };
2332
+ }
2333
+ function codeTraceBlockFromFinding(trajectoryId, source) {
2334
+ const parsed = CODE_TRACE_BLOCK_SUBJECT.exec(source.subject ?? "");
2335
+ if (!parsed) throw new Error(`CodeTraceBench model finding '${source.finding_id}' must set subject to incorrect-steps-<first>-<last>-<escaped|unescaped>-consequence-<step>, received '${source.subject ?? ""}'`);
2336
+ const firstStep = Number(parsed[1]);
2337
+ const lastStep = Number(parsed[2]);
2338
+ const consequenceStep = Number(parsed[4]);
2339
+ const cited = exactFindingSteps(trajectoryId, source);
2340
+ for (const step of cited) if (step < firstStep || step > lastStep) throw new Error(`CodeTraceBench model finding '${source.finding_id}' cites step ${step} outside its block ${firstStep}-${lastStep}`);
2341
+ return {
2342
+ firstStep,
2343
+ lastStep,
2344
+ consequenceStep,
2345
+ escapeStatus: parsed[3],
2346
+ severity: source.severity,
2347
+ claim: source.claim,
2348
+ confidence: source.confidence,
2349
+ ...source.rationale === void 0 ? {} : { rationale: source.rationale },
2350
+ ...source.recommended_action === void 0 ? {} : { recommendedAction: source.recommended_action },
2351
+ metadata: { sourceFindingId: source.finding_id }
2352
+ };
2353
+ }
2354
+ /**
2355
+ * Expand contiguous failure blocks into one scored finding per member step.
2356
+ *
2357
+ * The official scorer matches on area plus the exact step evidence URI, so
2358
+ * blocks never reach it: every runner reports blocks, and this function turns
2359
+ * them into the per-step findings the benchmark defines.
2360
+ */
2361
+ async function expandCodeTraceFailureBlocks(options) {
2362
+ const diagnostics = emptyCodeTraceBlockDiagnostics();
2363
+ diagnostics.reportedBlocks = options.blocks.length;
2364
+ if (options.blocks.length === 0) return {
2365
+ findings: [],
2366
+ diagnostics
2367
+ };
2368
+ assertCodeTraceBlockShape(options.blocks);
2369
+ const boundarySteps = options.blocks.flatMap((block) => [block.firstStep, block.lastStep]);
2370
+ const derivedSteps = options.blocks.flatMap((block) => [block.consequenceStep, ...interiorSteps(block)]);
2371
+ const evidenceByStep = await resolveAssistantStepEvidence({
2372
+ trajectoryId: options.trajectoryId,
2373
+ steps: boundarySteps,
2374
+ optionalSteps: derivedSteps,
2375
+ store: options.store,
2376
+ ...options.signal ? { signal: options.signal } : {}
2377
+ });
2378
+ const byStep = /* @__PURE__ */ new Map();
2379
+ for (const block of options.blocks) {
2380
+ if (!evidenceByStep.has(block.consequenceStep)) {
2381
+ diagnostics.blocksWithoutConsequenceEvidence.push(block);
2382
+ continue;
2383
+ }
2384
+ if (block.escapeStatus === "escaped") diagnostics.escapedBlocks += 1;
2385
+ for (let step = block.firstStep; step <= block.lastStep; step += 1) {
2386
+ if (!evidenceByStep.has(step)) {
2387
+ diagnostics.unresolvedBlockInteriorSteps.push(step);
2388
+ continue;
2389
+ }
2390
+ if (byStep.has(step)) {
2391
+ diagnostics.overlappingBlockSteps.push(step);
2392
+ continue;
2393
+ }
2394
+ byStep.set(step, block);
2395
+ }
2396
+ }
2397
+ return {
2398
+ findings: [...byStep].sort(([left], [right]) => left - right).map(([step, block]) => makeFinding({
2399
+ analyst_id: options.analystId,
2400
+ area: "incorrect",
2401
+ subject: `incorrect-step-${step}`,
2402
+ claim: `Step ${step} is incorrect. ${block.claim}`,
2403
+ rationale: block.rationale,
2404
+ severity: block.severity,
2405
+ confidence: block.confidence,
2406
+ evidence_refs: [evidenceByStep.get(step)],
2407
+ recommended_action: block.recommendedAction,
2408
+ metadata: {
2409
+ ...block.metadata,
2410
+ block_first_step: block.firstStep,
2411
+ block_last_step: block.lastStep,
2412
+ block_consequence_step: block.consequenceStep,
2413
+ escape_status: block.escapeStatus
2414
+ },
2415
+ ...options.producedAt === void 0 ? {} : { produced_at: options.producedAt },
2416
+ id_basis: `incorrect-step-${step}`
2417
+ })),
2418
+ diagnostics
2419
+ };
2420
+ }
2421
+ function assertCodeTraceBlockShape(blocks) {
2422
+ if (blocks.length > 16) throw new Error(`model reported ${blocks.length} failure blocks; the maximum is 16`);
2423
+ for (const block of blocks) {
2424
+ if (block.lastStep < block.firstStep) throw new Error(`failure block last_step ${block.lastStep} precedes first_step ${block.firstStep}`);
2425
+ const length = block.lastStep - block.firstStep + 1;
2426
+ if (length > 12) throw new Error(`failure block spans ${length} steps; the maximum is 12`);
2427
+ if (block.consequenceStep < block.firstStep) throw new Error(`failure block consequence_step ${block.consequenceStep} precedes first_step ${block.firstStep}`);
2428
+ }
2429
+ }
2430
+ function interiorSteps(block) {
2431
+ const steps = [];
2432
+ for (let step = block.firstStep + 1; step < block.lastStep; step += 1) steps.push(step);
2433
+ return steps;
2434
+ }
2435
+ function emptyCodeTraceBlockDiagnostics() {
2436
+ return {
2437
+ reportedBlocks: 0,
2438
+ escapedBlocks: 0,
2439
+ blocksWithoutConsequenceEvidence: [],
2440
+ unresolvedBlockInteriorSteps: [],
2441
+ overlappingBlockSteps: []
2442
+ };
2443
+ }
2444
+ function exactFindingSteps(trajectoryId, finding) {
2445
+ if (finding.evidence_refs.length === 0) throw new Error(`model finding '${finding.finding_id}' has no step evidence`);
2446
+ const steps = finding.evidence_refs.map((evidence) => {
2447
+ const parsed = codeTraceStepFromEvidence(evidence.uri);
2448
+ if (!parsed || parsed.traceId !== trajectoryId) throw new Error(`model finding '${finding.finding_id}' cites non-case evidence '${evidence.uri}'`);
2449
+ return parsed.step;
2450
+ });
2451
+ return [...new Set(steps)];
2452
+ }
2453
+ //#endregion
2454
+ //#region src/analyst/benchmark-public-types.ts
2455
+ function requiredString(value, field) {
2456
+ const trimmed = value.trim();
2457
+ if (!trimmed) throw new TypeError(`${field} must be a non-empty string`);
2458
+ return trimmed;
2459
+ }
2460
+ function positiveSafeInteger(value, field) {
2461
+ if (!Number.isSafeInteger(value) || value < 1) throw new RangeError(`${field} must be a positive safe integer`);
2462
+ return value;
2463
+ }
2464
+ function safeInteger(value, field) {
2465
+ if (!Number.isSafeInteger(value)) throw new RangeError(`${field} must be a safe integer`);
2466
+ return value;
2467
+ }
2468
+ function isRecord(value) {
2469
+ return typeof value === "object" && value !== null && !Array.isArray(value);
2470
+ }
2471
+ //#endregion
2160
2472
  //#region src/analyst/benchmark-public-data.ts
2161
2473
  const DEFAULT_MAX_PUBLIC_BENCHMARK_LABEL_BYTES = 256 * 1024 * 1024;
2162
2474
  const INPUT_OPEN_FLAGS = constants.O_RDONLY | (constants.O_NOFOLLOW ?? 0);
@@ -2655,15 +2967,6 @@ function fileContext() {
2655
2967
  }
2656
2968
  //#endregion
2657
2969
  //#region src/analyst/benchmark-public-model.ts
2658
- const TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS = [
2659
- 4096,
2660
- 2048,
2661
- 1024,
2662
- 512,
2663
- 256,
2664
- 128,
2665
- 64
2666
- ];
2667
2970
  /** One-shot JSON baseline. This is not a recursive trace analyst. */
2668
2971
  function createPublicBenchmarkDirectRunner(dataset, config) {
2669
2972
  const model = requiredString(config.model, "model");
@@ -2677,7 +2980,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
2677
2980
  responseCacheDir: requiredString(config.durability.responseCacheDir, "durability.responseCacheDir")
2678
2981
  } : void 0;
2679
2982
  const actor = dataset === "agentrx" ? "agentrx-root-cause-localizer" : "codetracebench-step-localizer";
2680
- const outputAdapter = dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-step";
2983
+ const outputAdapter = dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-block";
2681
2984
  const llmOptions = {
2682
2985
  baseUrl,
2683
2986
  apiKey,
@@ -2697,6 +3000,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
2697
3000
  benchmarkRepetition: String(context.repetition)
2698
3001
  };
2699
3002
  let rawPredictions = [];
3003
+ let rejectedBlocks = [];
2700
3004
  let modelFindings = [];
2701
3005
  let providerModel = model;
2702
3006
  let producedAt;
@@ -2748,6 +3052,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
2748
3052
  };
2749
3053
  const response = parsePublicBenchmarkModelResponse(dataset, cached.response);
2750
3054
  rawPredictions = response.findings;
3055
+ rejectedBlocks = response.rejectedBlocks;
2751
3056
  providerModel = cached.metadata.providerModel;
2752
3057
  producedAt = cached.metadata.producedAt;
2753
3058
  modelMetadata = {
@@ -2784,7 +3089,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
2784
3089
  ...cacheIdentity,
2785
3090
  callId: providerCallId,
2786
3091
  status: "succeeded",
2787
- response,
3092
+ response: completed.value,
2788
3093
  metadata: {
2789
3094
  providerModel: completed.result.model,
2790
3095
  providerDurationMs: completed.result.durationMs,
@@ -2816,6 +3121,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
2816
3121
  if (!paid.succeeded) throw paid.error;
2817
3122
  const response = paid.value.response;
2818
3123
  rawPredictions = response.findings;
3124
+ rejectedBlocks = response.rejectedBlocks;
2819
3125
  providerModel = paid.value.result.model;
2820
3126
  producedAt = paid.value.producedAt;
2821
3127
  modelMetadata = {
@@ -2828,7 +3134,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
2828
3134
  cost: costReceiptMetadata(paid.receipt)
2829
3135
  };
2830
3136
  }
2831
- modelFindings = await publicBenchmarkPredictionsToFindings({
3137
+ const converted = await publicBenchmarkPredictionsToFindings({
2832
3138
  dataset,
2833
3139
  trajectoryId,
2834
3140
  predictions: rawPredictions,
@@ -2838,6 +3144,14 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
2838
3144
  producedAt: requiredString(producedAt ?? "", "finding producedAt"),
2839
3145
  ...context.signal ? { signal: context.signal } : {}
2840
3146
  });
3147
+ modelFindings = converted.findings;
3148
+ if (converted.diagnostics) modelMetadata = {
3149
+ ...modelMetadata,
3150
+ blockDiagnostics: {
3151
+ ...converted.diagnostics,
3152
+ rejectedBlocks
3153
+ }
3154
+ };
2841
3155
  if (dataset === "codetracebench") await validateCodeTraceFindingEvidence({
2842
3156
  trajectoryId,
2843
3157
  findings: modelFindings,
@@ -2935,7 +3249,7 @@ const ModelSeveritySchema = z.enum([
2935
3249
  "low",
2936
3250
  "info"
2937
3251
  ]);
2938
- const PublicBenchmarkPredictionSchema = z.object({
3252
+ const AgentRxPredictionSchema = z.object({
2939
3253
  step: z.number().int().positive(),
2940
3254
  severity: ModelSeveritySchema,
2941
3255
  claim: z.string().min(1),
@@ -2943,6 +3257,34 @@ const PublicBenchmarkPredictionSchema = z.object({
2943
3257
  rationale: z.string().min(1).optional(),
2944
3258
  recommended_action: z.string().min(1).optional()
2945
3259
  }).strict();
3260
+ const CodeTraceBlockPredictionSchema = z.object({
3261
+ first_step: z.number().int().positive(),
3262
+ last_step: z.number().int().positive(),
3263
+ consequence_step: z.number().int().positive(),
3264
+ escape_status: z.enum(["escaped", "unescaped"]),
3265
+ severity: ModelSeveritySchema,
3266
+ claim: z.string().min(1),
3267
+ confidence: z.number().min(0).max(1),
3268
+ rationale: z.string().min(1).optional(),
3269
+ recommended_action: z.string().min(1).optional()
3270
+ }).strict().superRefine((block, ctx) => {
3271
+ if (block.last_step < block.first_step) {
3272
+ ctx.addIssue({
3273
+ code: z.ZodIssueCode.custom,
3274
+ message: `failure block last_step ${block.last_step} precedes first_step ${block.first_step}`
3275
+ });
3276
+ return;
3277
+ }
3278
+ const length = block.last_step - block.first_step + 1;
3279
+ if (length > 12) ctx.addIssue({
3280
+ code: z.ZodIssueCode.custom,
3281
+ message: `failure block spans ${length} steps; the maximum is 12`
3282
+ });
3283
+ if (block.consequence_step < block.first_step) ctx.addIssue({
3284
+ code: z.ZodIssueCode.custom,
3285
+ message: `failure block consequence_step ${block.consequence_step} precedes first_step ${block.first_step}`
3286
+ });
3287
+ });
2946
3288
  const AgentRxCategorySchema = z.enum([
2947
3289
  "instruction-plan-adherence-failure",
2948
3290
  "invention-of-new-information",
@@ -2955,28 +3297,52 @@ const AgentRxCategorySchema = z.enum([
2955
3297
  "system-failure",
2956
3298
  "inconclusive"
2957
3299
  ]);
2958
- const CodeTraceModelResponseSchema = z.object({
3300
+ const CodeTraceModelResponseEnvelopeSchema = z.object({
2959
3301
  report: z.string().min(1).max(4e3),
2960
- findings: z.array(PublicBenchmarkPredictionSchema).max(200)
3302
+ findings: z.array(z.unknown()).max(16)
2961
3303
  }).strict();
2962
3304
  const AgentRxModelResponseSchema = z.object({
2963
3305
  report: z.string().min(1).max(4e3),
2964
- findings: z.array(PublicBenchmarkPredictionSchema.extend({ category: AgentRxCategorySchema }).strict()).max(1)
3306
+ findings: z.array(AgentRxPredictionSchema.extend({ category: AgentRxCategorySchema }).strict()).max(1)
2965
3307
  }).strict();
2966
3308
  function parsePublicBenchmarkModelResponse(dataset, value) {
2967
- return dataset === "agentrx" ? AgentRxModelResponseSchema.parse(value) : CodeTraceModelResponseSchema.parse(value);
3309
+ if (dataset === "agentrx") return {
3310
+ ...AgentRxModelResponseSchema.parse(value),
3311
+ rejectedBlocks: []
3312
+ };
3313
+ const envelope = CodeTraceModelResponseEnvelopeSchema.parse(value);
3314
+ const findings = [];
3315
+ const rejectedBlocks = [];
3316
+ for (const [index, block] of envelope.findings.entries()) {
3317
+ const parsed = CodeTraceBlockPredictionSchema.safeParse(block);
3318
+ if (parsed.success) {
3319
+ findings.push(parsed.data);
3320
+ continue;
3321
+ }
3322
+ rejectedBlocks.push(`block ${index}: ${parsed.error.issues.map((issue) => `${issue.path.join(".") || "<root>"} ${issue.message}`).join("; ")}`);
3323
+ }
3324
+ if (findings.length === 0 && envelope.findings.length > 0) throw new ValidationError(`every reported failure block was malformed: ${rejectedBlocks.join(" | ")}`);
3325
+ return {
3326
+ report: envelope.report,
3327
+ findings,
3328
+ rejectedBlocks
3329
+ };
2968
3330
  }
2969
3331
  async function publicBenchmarkPredictionsToFindings(options) {
2970
- if (options.predictions.length === 0) return [];
2971
- const evidenceByStep = await resolveAssistantStepEvidence({
2972
- trajectoryId: options.trajectoryId,
2973
- steps: options.predictions.map((prediction) => prediction.step),
2974
- store: options.store,
2975
- ...options.signal ? { signal: options.signal } : {}
2976
- });
3332
+ if (options.predictions.length === 0 && options.dataset === "agentrx") return {
3333
+ findings: [],
3334
+ diagnostics: void 0
3335
+ };
2977
3336
  if (options.dataset === "agentrx") {
2978
3337
  const prediction = options.predictions[0];
3338
+ if (!("step" in prediction)) throw new Error("AgentRx model output must name a single root-cause step");
2979
3339
  if (!prediction.category) throw new Error("AgentRx model output is missing its failure category");
3340
+ const evidenceByStep = await resolveAssistantStepEvidence({
3341
+ trajectoryId: options.trajectoryId,
3342
+ steps: [prediction.step],
3343
+ store: options.store,
3344
+ ...options.signal ? { signal: options.signal } : {}
3345
+ });
2980
3346
  const [finding] = agentRxPredictionsToFindings(options.trajectoryId, [{
2981
3347
  failure_case: prediction.category,
2982
3348
  step_number: prediction.step,
@@ -2987,69 +3353,44 @@ async function publicBenchmarkPredictionsToFindings(options) {
2987
3353
  confidence: prediction.confidence
2988
3354
  });
2989
3355
  if (!finding) throw new Error("AgentRx output adapter produced no root-cause finding");
2990
- return [{
2991
- ...finding,
2992
- evidence_refs: [evidenceByStep.get(prediction.step)],
3356
+ return {
3357
+ findings: [{
3358
+ ...finding,
3359
+ evidence_refs: [evidenceByStep.get(prediction.step)],
3360
+ metadata: {
3361
+ ...finding.metadata,
3362
+ model: options.providerModel
3363
+ }
3364
+ }],
3365
+ diagnostics: void 0
3366
+ };
3367
+ }
3368
+ const blocks = options.predictions.map((prediction) => {
3369
+ if (!("first_step" in prediction)) throw new Error("CodeTraceBench model output must report first_step/last_step failure blocks");
3370
+ return {
3371
+ firstStep: prediction.first_step,
3372
+ lastStep: prediction.last_step,
3373
+ consequenceStep: prediction.consequence_step,
3374
+ escapeStatus: prediction.escape_status,
3375
+ severity: prediction.severity,
3376
+ claim: prediction.claim,
3377
+ confidence: prediction.confidence,
3378
+ ...prediction.rationale === void 0 ? {} : { rationale: prediction.rationale },
3379
+ ...prediction.recommended_action === void 0 ? {} : { recommendedAction: prediction.recommended_action },
2993
3380
  metadata: {
2994
- ...finding.metadata,
3381
+ analysis_mode: "direct-baseline",
2995
3382
  model: options.providerModel
2996
3383
  }
2997
- }];
2998
- }
2999
- const byStep = /* @__PURE__ */ new Map();
3000
- for (const prediction of options.predictions) if (!byStep.has(prediction.step)) byStep.set(prediction.step, prediction);
3001
- return [...byStep].sort(([left], [right]) => left - right).map(([step, prediction]) => makeFinding({
3002
- analyst_id: options.analystId,
3003
- area: "incorrect",
3004
- subject: `incorrect-step-${step}`,
3005
- claim: `Step ${step} is incorrect. ${prediction.claim}`,
3006
- rationale: prediction.rationale,
3007
- severity: prediction.severity,
3008
- confidence: prediction.confidence,
3009
- evidence_refs: [evidenceByStep.get(step)],
3010
- recommended_action: prediction.recommended_action,
3011
- metadata: {
3012
- analysis_mode: "direct-baseline",
3013
- model: options.providerModel
3014
- },
3015
- produced_at: options.producedAt,
3016
- id_basis: `incorrect-step-${step}`
3017
- }));
3018
- }
3019
- function publicBenchmarkSystemPrompt(dataset) {
3020
- return `${dataset === "agentrx" ? AGENT_RX_PROMPT : CODE_TRACE_BENCH_ANALYST_PROMPT}
3021
-
3022
- Each finding must contain only:
3023
- - "step": a positive integer matching an existing assistant LLM span named step-<n>
3024
- - "severity": "critical", "high", "medium", "low", or "info"
3025
- - "claim": one sentence
3026
- - "confidence": a number from 0 through 1
3027
- - optional "rationale" and "recommended_action" strings
3028
- ${dataset === "agentrx" ? `- "category": one allowed failure category listed above` : ""}
3029
-
3030
- Return exactly one JSON object with:
3031
- - "report": a concise evidence-based explanation, at most 4000 characters
3032
- - "findings": the strict finding array
3033
- Use an empty findings array when the trace does not support a finding.
3034
- Do not return a bare array, markdown, trace URIs, copied excerpts, or fields not listed above.
3035
- The runner constructs exact trace URIs and action previews from each selected step.`;
3036
- }
3037
- function publicBenchmarkProtocolSha256(dataset) {
3038
- return sha256Digest(JSON.stringify({
3039
- dataset,
3040
- systemPrompt: publicBenchmarkSystemPrompt(dataset),
3041
- transport: {
3042
- attempts: 1,
3043
- jsonMode: true,
3044
- thinking: "disabled"
3045
- },
3046
- traceProjectionAttributeByteCaps: TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS,
3047
- evidence: {
3048
- location: "model-selected-positive-integer-assistant-step",
3049
- uri: "deterministic-trace-uri",
3050
- excerpt: `exact-action-prefix-512`
3051
- }
3052
- }));
3384
+ };
3385
+ });
3386
+ return expandCodeTraceFailureBlocks({
3387
+ trajectoryId: options.trajectoryId,
3388
+ blocks,
3389
+ store: options.store,
3390
+ analystId: options.analystId,
3391
+ producedAt: options.producedAt,
3392
+ ...options.signal ? { signal: options.signal } : {}
3393
+ });
3053
3394
  }
3054
3395
  async function prepareSingleTraceContext(store, context) {
3055
3396
  const storeContext = context.signal ? { signal: context.signal } : void 0;
@@ -3069,40 +3410,6 @@ async function prepareSingleTraceContext(store, context) {
3069
3410
  });
3070
3411
  }
3071
3412
  }
3072
- const AGENT_RX_PROMPT = `Analyze exactly one failed agent trajectory.
3073
- Find the first unrecoverable critical failure, not every later symptom.
3074
- Inspect the complete supplied trace data.
3075
- Emit zero findings only when the trace does not contain enough evidence.
3076
- Otherwise emit exactly one finding.
3077
- Its category MUST be exactly one of:
3078
- instruction-plan-adherence-failure
3079
- invention-of-new-information
3080
- invalid-invocation
3081
- misinterpretation-of-tool-output-handoff-failure
3082
- intent-plan-misalignment
3083
- underspecified-user-intent
3084
- intent-not-supported
3085
- guardrails-triggered
3086
- system-failure
3087
- inconclusive
3088
- Its step is the positive integer n from the first unrecoverable assistant span named step-<n>.`;
3089
- const CODE_TRACE_BENCH_ANALYST_PROMPT = `Analyze exactly one coding-agent trajectory and its attached final verification.
3090
- Your task is the CodeTraceBench incorrect-step task: identify every wrong state-changing action, bad hypothesis that drives an action, and regression.
3091
- An incorrect step remains incorrect when the agent later recovers or the final verification passes.
3092
- Inspect the complete supplied trace data.
3093
- Use the final-verification outcome as evidence about the final state, not as a rule for whether earlier steps were incorrect.
3094
- For each candidate, inspect the assistant action and its following observation.
3095
- Label a failed command when the assistant caused it through a wrong action or unsupported hypothesis.
3096
- Label the later corrective action only when that action is itself wrong.
3097
- Do not label a diagnostic probe merely because it exposes an earlier defect.
3098
- Do not label a redundant but correct read or search; CodeTraceBench scores unuseful steps separately, and this run scores incorrect steps only.
3099
- Do not label a step solely because final verification failed.
3100
- When final verification is unavailable, use only directly observed trajectory evidence.
3101
- Emit one finding per incorrect assistant step.
3102
- Each finding's step MUST be the positive integer n from an existing assistant LLM span named step-<n>.
3103
- Never select an EVALUATOR, TOOL, CHAIN, final-verification, benchmark-verification, or message-<n> span.
3104
- Before emitting a finding, inspect its candidate span's attributes.content and describe only the action shown there.
3105
- When the trajectory has no incorrect steps, return an empty findings array.`;
3106
3413
  function trajectoryIdFromCaseId$1(dataset, caseId) {
3107
3414
  const prefix = dataset === "agentrx" ? "agentrx:" : "codetrace:";
3108
3415
  if (!caseId.startsWith(prefix) || caseId.length === prefix.length) throw new Error(`unexpected ${dataset} benchmark case id '${caseId}'`);
@@ -3110,41 +3417,13 @@ function trajectoryIdFromCaseId$1(dataset, caseId) {
3110
3417
  }
3111
3418
  //#endregion
3112
3419
  //#region src/analyst/benchmark-public-rlm.ts
3113
- const AGENT_RX_RLM_INSTRUCTIONS = `Analyze exactly one failed agent trajectory.
3114
- Find the first unrecoverable critical failure, not every later symptom.
3115
- Use the trace tools to inspect the action and its following observation.
3116
- Emit zero findings only when evidence does not support a root cause.
3117
- Otherwise emit exactly one finding whose subject is exactly one of:
3118
- instruction-plan-adherence-failure
3119
- invention-of-new-information
3120
- invalid-invocation
3121
- misinterpretation-of-tool-output-handoff-failure
3122
- intent-plan-misalignment
3123
- underspecified-user-intent
3124
- intent-not-supported
3125
- guardrails-triggered
3126
- system-failure
3127
- inconclusive
3128
- Cite exactly one assistant span named step-<n> as trace://<trace-id>/span/step-<n>.
3129
- The excerpt must quote the assistant action exactly.`;
3130
- const CODE_TRACE_RLM_INSTRUCTIONS = `${CODE_TRACE_BENCH_ANALYST_PROMPT}
3131
- Use the trace tools rather than asking for the whole trajectory in the prompt.
3132
- Keep retrieved trace objects in Python variables.
3133
- Never print an entire trace, full source file, or more than 12000 characters in one iteration.
3134
- Build a compact table of assistant step ids, actions, following observations, and final verification.
3135
- Inspect suspicious steps with viewSpans or searchSpan instead of repeatedly printing the table.
3136
- Submit as soon as every state-changing assistant step has a supported verdict.
3137
- For each incorrect step, set subject to incorrect-step-<n>.
3138
- Cite the assistant span as trace://<URL-encoded-trace-id>/span/step-<n>.
3139
- The evidence excerpt must be an exact quote from that span's action content.
3140
- Return no finding for a clean trajectory.`;
3141
3420
  /** Public benchmark candidate that runs the actual recursive trace analyst. */
3142
3421
  function createPublicBenchmarkRlmRunner(dataset, config) {
3143
3422
  const costLedger = config.costLedger ?? new CostLedger();
3144
3423
  const limits = {
3145
- maxIterations: config.dspyRlm?.maxIterations ?? 8,
3146
- maxLlmCalls: config.dspyRlm?.maxLlmCalls ?? 4,
3147
- maxToolCalls: config.dspyRlm?.maxToolCalls ?? 32,
3424
+ maxIterations: config.dspyRlm?.maxIterations ?? 14,
3425
+ maxLlmCalls: config.dspyRlm?.maxLlmCalls ?? 8,
3426
+ maxToolCalls: config.dspyRlm?.maxToolCalls ?? 80,
3148
3427
  maxOutputChars: config.dspyRlm?.maxOutputChars ?? 8e3
3149
3428
  };
3150
3429
  const pricing = config.pricing ?? pricingForModel(config.model);
@@ -3205,20 +3484,22 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
3205
3484
  },
3206
3485
  produced_at: producedAt
3207
3486
  }));
3208
- const findings = adaptPublicBenchmarkFindings(dataset, trajectoryId, rawFindings, "dspy-rlm");
3209
- if (dataset === "codetracebench") await validateCodeTraceFindingEvidence({
3487
+ const adapted = await adaptPublicBenchmarkFindings({
3488
+ dataset,
3210
3489
  trajectoryId,
3211
- findings,
3490
+ findings: rawFindings,
3491
+ analystId: "dspy-rlm",
3212
3492
  store: input.traceStore,
3213
3493
  ...context.signal ? { signal: context.signal } : {}
3214
3494
  });
3215
3495
  return {
3216
- findings,
3496
+ findings: adapted.findings,
3217
3497
  usage,
3218
3498
  metadata: {
3219
3499
  analysisMode: "recursive",
3220
3500
  engine: "dspy-rlm",
3221
3501
  protocolSha256: publicBenchmarkProtocolSha256(dataset),
3502
+ ...adapted.diagnostics ? { blockDiagnostics: adapted.diagnostics } : {},
3222
3503
  answer: completed.answer,
3223
3504
  trajectory: completed.trajectory,
3224
3505
  modelCalls: completed.modelCalls,
@@ -3250,7 +3531,7 @@ function publicBenchmarkDefinition(dataset, limits) {
3250
3531
  area: dataset === "agentrx" ? "root-cause" : "incorrect",
3251
3532
  version: "1.0.0",
3252
3533
  question: dataset === "agentrx" ? "What is the first unrecoverable root cause in this failed trajectory?" : "Which assistant steps are incorrect under the CodeTraceBench definition?",
3253
- instructions: dataset === "agentrx" ? AGENT_RX_RLM_INSTRUCTIONS : CODE_TRACE_RLM_INSTRUCTIONS,
3534
+ instructions: publicBenchmarkRlmInstructions(dataset),
3254
3535
  toolGroup: "singleTrace",
3255
3536
  limits
3256
3537
  };
@@ -3341,7 +3622,7 @@ function createRunIdentity(config, prepared) {
3341
3622
  analystProtocolSha256: publicBenchmarkProtocolSha256(config.dataset),
3342
3623
  implementationSha256: ANALYST_BENCHMARK_IMPLEMENTATION_SHA256,
3343
3624
  dependencyLockSha256: ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256,
3344
- runnerIds: ["empty", "dspy-rlm"]
3625
+ runnerIds: ["empty", config.analyst]
3345
3626
  },
3346
3627
  inputs: {
3347
3628
  labelsSha256: prepared.labelsSha256,
@@ -3512,7 +3793,7 @@ function createObservationAppender(path, runIdentitySha256, progress) {
3512
3793
  return write;
3513
3794
  };
3514
3795
  }
3515
- async function readProgress(path, runIdentitySha256, caseIds, repetitions) {
3796
+ async function readProgress(path, runIdentitySha256, caseIds, repetitions, analystRunnerId) {
3516
3797
  const rawLines = (await readRegularFile(path, "benchmark observation log")).split("\n");
3517
3798
  if (rawLines.at(-1) === "") rawLines.pop();
3518
3799
  const observations = [];
@@ -3546,7 +3827,7 @@ async function readProgress(path, runIdentitySha256, caseIds, repetitions) {
3546
3827
  previousRowSha256: parsed.previousRowSha256,
3547
3828
  observation
3548
3829
  }) !== parsed.rowSha256) throw new Error(`benchmark observation row ${index + 1} digest does not match its contents`);
3549
- if (!allowedCases.has(observation.caseId) || observation.runnerId !== "empty" && observation.runnerId !== "dspy-rlm" || observation.repetition >= repetitions || observation.executionIndex >= plannedObservationCount) throw new Error(`benchmark observation row ${index + 1} does not match a planned case, runner, and repetition`);
3830
+ if (!allowedCases.has(observation.caseId) || observation.runnerId !== "empty" && observation.runnerId !== analystRunnerId || observation.repetition >= repetitions || observation.executionIndex >= plannedObservationCount) throw new Error(`benchmark observation row ${index + 1} does not match a planned case, runner, and repetition`);
3550
3831
  if (executionIndexes.has(observation.executionIndex)) throw new Error(`duplicate benchmark executionIndex ${observation.executionIndex} at line ${index + 1}`);
3551
3832
  observations.push(observation);
3552
3833
  seen.add(key);
@@ -3991,7 +4272,7 @@ function assertCompletedArtifactMatchesRun(artifact, manifest, observations, pre
3991
4272
  if (canonicalJson(artifact.inputs) !== canonicalJson(expectedInputs)) throw new Error("completed benchmark result inputs do not match the run manifest");
3992
4273
  const provenance = artifact.result.provenance;
3993
4274
  const expectedDatasetId = config.dataset === "agentrx" ? "microsoft/AgentRx" : "NJU-LINK/CodeTraceBench";
3994
- const expectedOutputAdapter = config.dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-step";
4275
+ const expectedOutputAdapter = config.dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-block";
3995
4276
  if (provenance.id !== `${config.dataset}-real-model-analyst` || provenance.startedAt !== manifest.createdAt || !Number.isFinite(Date.parse(provenance.endedAt)) || Date.parse(provenance.endedAt) < Date.parse(provenance.startedAt) || canonicalJson(provenance.dataset) !== canonicalJson({
3996
4277
  id: expectedDatasetId,
3997
4278
  revision: config.datasetRevision,
@@ -4001,7 +4282,7 @@ function assertCompletedArtifactMatchesRun(artifact, manifest, observations, pre
4001
4282
  if (canonicalJson(artifact.result.summaries) !== canonicalJson(expectedSummaries)) throw new Error("completed benchmark summaries do not match durable observations");
4002
4283
  const expectedComparisons = [compareAnalystRunners(artifact.result, {
4003
4284
  baselineRunnerId: "empty",
4004
- candidateRunnerId: "dspy-rlm",
4285
+ candidateRunnerId: config.runnerIds[1],
4005
4286
  seed: config.seed
4006
4287
  })];
4007
4288
  if (canonicalJson(artifact.comparisons) !== canonicalJson(expectedComparisons)) throw new Error("completed benchmark comparisons do not match durable observations");
@@ -4134,7 +4415,7 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
4134
4415
  const localIdentitySha256 = digestCanonical(localReceipt.local);
4135
4416
  const identitySha256 = digestCanonical(identity);
4136
4417
  const manifest = config.resume ? await readAndValidateResumeFiles(paths, identity, identitySha256, localIdentitySha256, localReceipt) : await initializeRunFiles(paths, identity, identitySha256, localIdentitySha256, localReceipt);
4137
- const progress = await readProgress(paths.observations, manifest.identitySha256, prepared.selectedCaseIds, config.repetitions);
4418
+ const progress = await readProgress(paths.observations, manifest.identitySha256, prepared.selectedCaseIds, config.repetitions, config.analyst);
4138
4419
  const costLedger = createRunCostLedger({
4139
4420
  storage: fsCampaignStorage(),
4140
4421
  runDir: paths.directory,
@@ -4147,10 +4428,10 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
4147
4428
  const markdown = renderArtifactMarkdown(artifact);
4148
4429
  await writeExclusiveOrVerify(paths.report, markdown);
4149
4430
  printSuccessSummary(artifact, paths);
4150
- return benchmarkExitCode(artifact.result);
4431
+ return benchmarkExitCode(artifact.result, config.analyst);
4151
4432
  }
4152
4433
  if (await regularFileExists(paths.report)) throw new Error(`benchmark report exists without a completed result; refusing ambiguous resume: ${paths.report}`);
4153
- const createAnalystRunner = dependencies.createAnalystRunner ?? ((dataset, model) => createPublicBenchmarkRlmRunner(dataset, model));
4434
+ const createAnalystRunner = dependencies.createAnalystRunner ?? ((dataset, model) => config.analyst === "direct" ? createPublicBenchmarkDirectRunner(dataset, model) : createPublicBenchmarkRlmRunner(dataset, model));
4154
4435
  const runners = [emptyPublicBenchmarkRunner(), createAnalystRunner(config.dataset, {
4155
4436
  ...config.model,
4156
4437
  costLedger,
@@ -4172,7 +4453,7 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
4172
4453
  initialObservations: progress.observations,
4173
4454
  signal: runAbort.signal,
4174
4455
  onObservation: async (observation) => {
4175
- assertObservationAccountingComplete(observation, costLedger);
4456
+ assertObservationAccountingComplete(observation, costLedger, config.analyst);
4176
4457
  await appendObservation(observation);
4177
4458
  },
4178
4459
  resolveEvidence: traceStoreEvidenceResolver((input) => {
@@ -4193,7 +4474,7 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
4193
4474
  },
4194
4475
  metadata: {
4195
4476
  model: config.model.model,
4196
- outputAdapter: config.dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-step",
4477
+ outputAdapter: config.dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-block",
4197
4478
  caseSelection: prepared.selection.method,
4198
4479
  caseSelectionSeed: config.seed,
4199
4480
  selectionStratified: prepared.selection.stratified,
@@ -4211,11 +4492,11 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
4211
4492
  }
4212
4493
  assertCostLedgerFinalizable(costLedger);
4213
4494
  result.provenance.startedAt = manifest.createdAt;
4214
- const persisted = await readProgress(paths.observations, manifest.identitySha256, prepared.selectedCaseIds, config.repetitions);
4495
+ const persisted = await readProgress(paths.observations, manifest.identitySha256, prepared.selectedCaseIds, config.repetitions, config.analyst);
4215
4496
  assertSameObservations(result.observations, persisted.observations);
4216
4497
  const comparisons = [compareAnalystRunners(result, {
4217
4498
  baselineRunnerId: "empty",
4218
- candidateRunnerId: "dspy-rlm",
4499
+ candidateRunnerId: config.analyst,
4219
4500
  seed: config.seed
4220
4501
  })];
4221
4502
  const codeTraceCalibration = config.dataset === "codetracebench" ? summarizeCodeTraceCalibration(result) : void 0;
@@ -4260,7 +4541,7 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
4260
4541
  await writeExclusiveOrVerify(paths.result, `${JSON.stringify(artifact, null, 2)}\n`);
4261
4542
  await writeExclusiveOrVerify(paths.report, markdown);
4262
4543
  printSuccessSummary(artifact, paths);
4263
- return benchmarkExitCode(result);
4544
+ return benchmarkExitCode(result, config.analyst);
4264
4545
  }
4265
4546
  const NON_SCORABLE_COST_ERRORS = /* @__PURE__ */ new Set([
4266
4547
  "CostAccountingIncompleteError",
@@ -4270,27 +4551,33 @@ const NON_SCORABLE_COST_ERRORS = /* @__PURE__ */ new Set([
4270
4551
  "CostReceiptCaptureError",
4271
4552
  "CostReservationExceededError"
4272
4553
  ]);
4273
- function assertObservationAccountingComplete(observation, costLedger) {
4554
+ function assertObservationAccountingComplete(observation, costLedger, analystRunnerId) {
4274
4555
  if (observation.error && NON_SCORABLE_COST_ERRORS.has(observation.error.class)) throw new CostAccountingIncompleteError(`Analyst benchmark stopped before scoring: ${observation.error.message}`);
4275
- if (observation.runnerId !== "dspy-rlm") return;
4276
- const summary = costLedger.summary({
4277
- channel: "analyst",
4278
- tags: {
4279
- benchmarkCaseId: observation.caseId,
4280
- benchmarkRepetition: String(observation.repetition)
4281
- }
4282
- });
4283
- if (!summary.accountingComplete || summary.pendingCalls > 0 || summary.unresolvedCalls > 0) throw accountingError(costLedger, "the recursive analyst has incomplete cost accounting", {
4556
+ if (observation.runnerId !== analystRunnerId) return;
4557
+ const filter = {
4284
4558
  channel: "analyst",
4285
4559
  tags: {
4286
4560
  benchmarkCaseId: observation.caseId,
4287
4561
  benchmarkRepetition: String(observation.repetition)
4288
4562
  }
4289
- });
4563
+ };
4564
+ if (!costAccountingIsTrustworthy(costLedger.summary(filter))) throw accountingError(costLedger, "the recursive analyst has incomplete cost accounting", filter);
4565
+ }
4566
+ const BUDGET_BREACH_REASON = /exceeding its enforced maximum/;
4567
+ /**
4568
+ * Cost accounting is trustworthy when every call resolved and none breached its
4569
+ * budget. A recursive analyst on a real provider will occasionally receive a
4570
+ * settled response whose usage the provider omitted; that call is honestly
4571
+ * recorded as unknown and excluded from the reported cost, so it does not
4572
+ * invalidate a completed run. A call left pending, one lost, or one charged
4573
+ * beyond its maximum is a genuine integrity failure and still halts.
4574
+ */
4575
+ function costAccountingIsTrustworthy(summary) {
4576
+ if (summary.pendingCalls > 0 || summary.unresolvedCalls > 0) return false;
4577
+ return !summary.incompleteReasons.some((reason) => BUDGET_BREACH_REASON.test(reason));
4290
4578
  }
4291
4579
  function assertCostLedgerFinalizable(costLedger) {
4292
- const summary = costLedger.summary();
4293
- if (!summary.accountingComplete || summary.pendingCalls > 0 || summary.unresolvedCalls > 0) throw accountingError(costLedger, "the run has pending or incomplete cost entries");
4580
+ if (!costAccountingIsTrustworthy(costLedger.summary())) throw accountingError(costLedger, "the run has pending or budget-breaching cost entries");
4294
4581
  }
4295
4582
  function accountingError(costLedger, reason, filter) {
4296
4583
  const details = costLedger.summary(filter).incompleteReasons.slice(0, 3).join("; ");
@@ -4302,6 +4589,9 @@ Run the recursive DSPy trace analyst against public AgentRx or CodeTraceBench la
4302
4589
 
4303
4590
  Required:
4304
4591
  --dataset agentrx|codetracebench
4592
+ --analyst dspy-rlm|direct Scored analyst. Default: dspy-rlm.
4593
+ 'direct' is the retired one-shot runner that
4594
+ produced the published evidence.
4305
4595
  --labels <dataset.json|dataset.jsonl>
4306
4596
  --trace-dir <one-trace-per-file OTLP JSONL directory>
4307
4597
  --artifact-dir <extracted artifact root> Required for CodeTraceBench
@@ -4318,7 +4608,7 @@ Controls:
4318
4608
  --seed <integer> Case-selection and comparison seed. Default: 0
4319
4609
  --concurrency <positive integer> Parallel benchmark jobs. Default: 1
4320
4610
  --repetitions <positive integer> Runs per case and runner. Default: 1
4321
- --max-output-tokens <positive> Model output limit per call. Default: 4096
4611
+ --max-output-tokens <positive> Model output limit per call. Default: 16384
4322
4612
  --python <executable> Python with agent-eval-rpc[dspy]. Default: python
4323
4613
  --timeout-ms <positive> Model analyst deadline per case. Default: 300000
4324
4614
  --max-cost-usd <positive> Run-wide spend limit. Default: 5
@@ -4344,8 +4634,11 @@ function parseCommandConfig(argv, env) {
4344
4634
  if (!apiKey) throw new Error(`--api-key-env points to an empty or missing variable: ${apiKeyEnv}`);
4345
4635
  const maxCostUsd = positiveFiniteFlag(flags, "max-cost-usd", 5);
4346
4636
  const python = flags.get("python")?.trim();
4637
+ const analyst = flags.get("analyst")?.trim() ?? "dspy-rlm";
4638
+ if (analyst !== "dspy-rlm" && analyst !== "direct") throw new Error("--analyst must be 'dspy-rlm' or 'direct'");
4347
4639
  return {
4348
4640
  dataset,
4641
+ analyst,
4349
4642
  labelsPath: requiredFlag(flags, "labels"),
4350
4643
  traceDir: requiredFlag(flags, "trace-dir"),
4351
4644
  ...artifactDir ? { artifactDir } : {},
@@ -4356,7 +4649,7 @@ function parseCommandConfig(argv, env) {
4356
4649
  baseUrl: openAiCompatibleBaseUrl(requiredFlag(flags, "base-url")),
4357
4650
  apiKey,
4358
4651
  model: requiredFlag(flags, "model"),
4359
- maxOutputTokens: positiveFlag(flags, "max-output-tokens", 4096),
4652
+ maxOutputTokens: positiveFlag(flags, "max-output-tokens", 16384),
4360
4653
  timeoutMs: positiveFlag(flags, "timeout-ms", 3e5),
4361
4654
  maxCostUsdPerAnalysis: maxCostUsd,
4362
4655
  ...python ? { dspyRlm: { runner: { command: python } } } : {}
@@ -4396,6 +4689,7 @@ function parseFlags(argv) {
4396
4689
  const KNOWN_FLAGS = /* @__PURE__ */ new Set([
4397
4690
  "resume",
4398
4691
  "dataset",
4692
+ "analyst",
4399
4693
  "labels",
4400
4694
  "trace-dir",
4401
4695
  "artifact-dir",
@@ -4519,8 +4813,8 @@ function renderArtifactMarkdown(artifact) {
4519
4813
  const verificationMarkdown = artifact.inputs.dataset === "codetracebench" ? `\n\n${renderVerificationAvailability(artifact.inputs.verificationAvailability)}` : "";
4520
4814
  return `${renderAnalystBenchmarkMarkdown(artifact.result, artifact.comparisons).trimEnd()}${calibrationMarkdown}${verificationMarkdown}\n\n${renderSelectionMarkdown(artifact.inputs.selection.report)}\n`;
4521
4815
  }
4522
- function benchmarkExitCode(result) {
4523
- return result.summaries.find((summary) => summary.runnerId === "dspy-rlm")?.failedRuns ? 2 : 0;
4816
+ function benchmarkExitCode(result, analystRunnerId) {
4817
+ return result.summaries.find((summary) => summary.runnerId === analystRunnerId)?.failedRuns ? 2 : 0;
4524
4818
  }
4525
4819
  function printSuccessSummary(artifact, paths) {
4526
4820
  const failures = artifact.result.summaries.reduce((total, summary) => total + summary.failedRuns, 0);
@@ -4532,6 +4826,6 @@ function shellQuote(value) {
4532
4826
  return /^[a-zA-Z0-9_./:@%+=,-]+$/.test(value) ? value : `'${value.replaceAll("'", "'\\''")}'`;
4533
4827
  }
4534
4828
  //#endregion
4535
- export { ANALYST_BENCHMARK_IMPLEMENTATION_FILES as A, summarizeAgentRxCalibration as B, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as C, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as D, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as E, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as F, normalizeAgentRxCategory as G, codeTracerPredictionsToFindings as H, ANALYST_BENCHMARK_MANIFEST_FILE as I, roundAgentRxStep as K, ANALYST_BENCHMARK_OBSERVATIONS_FILE as L, analystBenchmarkDependencyLockDigest as M, analystBenchmarkImplementationDigest as N, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as O, ANALYST_BENCHMARK_COST_LEDGER_FILE as P, AGENT_RX_UPSTREAM_REVISION as R, emptyPublicBenchmarkRunner as S, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as T, agentRxBenchmarkCase as U, codeTraceBenchCase as V, agentRxPredictionsToFindings as W, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as _, renderCodeTraceCalibrationMarkdown as a, parseVerificationOutcome as b, createPublicBenchmarkRlmRunner as c, publicBenchmarkProtocolSha256 as d, loadPublicBenchmarkRows as f, selectPublicBenchmarkRows as g, publicBenchmarkSelectionReport as h, readAnalystBenchmarkArtifact as i, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as j, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as k, CODE_TRACE_BENCH_ANALYST_PROMPT as l, publicBenchmarkDistributions as m, runAnalystBenchmarkCommand as n, summarizeCodeTraceCalibration as o, preparePublicAnalystBenchmark as p, normalizeBenchmarkLabel as q, renderAnalystBenchmarkMarkdown as r, compareAnalystRunners as s, ANALYST_BENCHMARK_HELP as t, createPublicBenchmarkDirectRunner as u, appendVerificationArtifactsToOtlp as v, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as w, adaptPublicBenchmarkFindings as x, loadCodeTraceVerificationArtifacts as y, renderAgentRxCalibrationMarkdown as z };
4829
+ export { ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as A, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as B, publicBenchmarkSystemPrompt as C, parseVerificationOutcome as D, loadCodeTraceVerificationArtifacts as E, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as F, summarizeAgentRxCalibration as G, ANALYST_BENCHMARK_OBSERVATIONS_FILE as H, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as I, agentRxBenchmarkCase as J, codeTraceBenchCase as K, analystBenchmarkDependencyLockDigest as L, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as M, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as N, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as O, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as P, normalizeBenchmarkLabel as Q, analystBenchmarkImplementationDigest as R, publicBenchmarkRlmInstructions as S, appendVerificationArtifactsToOtlp as T, AGENT_RX_UPSTREAM_REVISION as U, ANALYST_BENCHMARK_MANIFEST_FILE as V, renderAgentRxCalibrationMarkdown as W, normalizeAgentRxCategory as X, agentRxPredictionsToFindings as Y, roundAgentRxStep as Z, expandCodeTraceFailureBlocks as _, renderCodeTraceCalibrationMarkdown as a, MAX_INCORRECT_BLOCK_STEPS as b, createPublicBenchmarkRlmRunner as c, preparePublicAnalystBenchmark as d, publicBenchmarkDistributions as f, emptyPublicBenchmarkRunner as g, adaptPublicBenchmarkFindings as h, readAnalystBenchmarkArtifact as i, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as j, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as k, createPublicBenchmarkDirectRunner as l, selectPublicBenchmarkRows as m, runAnalystBenchmarkCommand as n, summarizeCodeTraceCalibration as o, publicBenchmarkSelectionReport as p, codeTracerPredictionsToFindings as q, renderAnalystBenchmarkMarkdown as r, compareAnalystRunners as s, ANALYST_BENCHMARK_HELP as t, loadPublicBenchmarkRows as u, CODE_TRACE_BENCH_ANALYST_PROMPT as v, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as w, publicBenchmarkProtocolSha256 as x, MAX_INCORRECT_BLOCKS as y, ANALYST_BENCHMARK_COST_LEDGER_FILE as z };
4536
4830
 
4537
- //# sourceMappingURL=benchmark-command-DzUZJl8M.js.map
4831
+ //# sourceMappingURL=benchmark-command-CK0UnXAD.js.map