@tangle-network/agent-eval 0.138.0 → 0.139.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (164) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/README.md +2 -1
  3. package/dist/analyst/index.d.ts +41 -94
  4. package/dist/analyst/index.d.ts.map +1 -1
  5. package/dist/analyst/index.js +9 -24
  6. package/dist/analyst/index.js.map +1 -1
  7. package/dist/{benchmark-D8dkki-J.js → benchmark-CYtcIF2V.js} +2 -2
  8. package/dist/{benchmark-D8dkki-J.js.map → benchmark-CYtcIF2V.js.map} +1 -1
  9. package/dist/{benchmark-DlQgU_XI.d.ts → benchmark-DDVdWcwA.d.ts} +3 -3
  10. package/dist/{benchmark-DlQgU_XI.d.ts.map → benchmark-DDVdWcwA.d.ts.map} +1 -1
  11. package/dist/{benchmark-command-CMqVqReF.js → benchmark-command-BKfjOBJ5.js} +243 -38
  12. package/dist/benchmark-command-BKfjOBJ5.js.map +1 -0
  13. package/dist/benchmarks/index.d.ts +1 -1
  14. package/dist/benchmarks/index.js +1 -1
  15. package/dist/{benchmarks-BJ_xK5rQ.js → benchmarks-zxhy1QV3.js} +4 -4
  16. package/dist/{benchmarks-BJ_xK5rQ.js.map → benchmarks-zxhy1QV3.js.map} +1 -1
  17. package/dist/campaign/index.d.ts +5 -5
  18. package/dist/campaign/index.js +4 -3
  19. package/dist/{campaign-BIBS-NHV.js → campaign-DrS6_hLd.js} +10 -9
  20. package/dist/campaign-DrS6_hLd.js.map +1 -0
  21. package/dist/canonical-D011XM8r.js +86 -0
  22. package/dist/canonical-D011XM8r.js.map +1 -0
  23. package/dist/cli.js +3 -3
  24. package/dist/{client-BwPKohkJ.d.ts → client-BohnDFBq.d.ts} +4 -4
  25. package/dist/{client-BwPKohkJ.d.ts.map → client-BohnDFBq.d.ts.map} +1 -1
  26. package/dist/{completion-verifier-B4-IMYcS.d.ts → completion-verifier-IPoP4fQO.d.ts} +178 -4
  27. package/dist/completion-verifier-IPoP4fQO.d.ts.map +1 -0
  28. package/dist/contract/index.d.ts +10 -10
  29. package/dist/contract/index.js +8 -7
  30. package/dist/contract/index.js.map +1 -1
  31. package/dist/control.d.ts +2 -2
  32. package/dist/{cost-ledger-CHDLA0Ss.js → cost-ledger-CZ9diLxY.js} +7 -7
  33. package/dist/cost-ledger-CZ9diLxY.js.map +1 -0
  34. package/dist/{cost-ledger-B1D3COAc.d.ts → cost-ledger-DKgyIWRj.d.ts} +5 -2
  35. package/dist/cost-ledger-DKgyIWRj.d.ts.map +1 -0
  36. package/dist/default-registry-B8vf7Rmf.d.ts +118 -0
  37. package/dist/default-registry-B8vf7Rmf.d.ts.map +1 -0
  38. package/dist/{default-registry-lp5R0lve.js → default-registry-BgJJItGr.js} +57 -1532
  39. package/dist/default-registry-BgJJItGr.js.map +1 -0
  40. package/dist/dspy-rlm-engine-DTkVyDX-.js +344 -0
  41. package/dist/dspy-rlm-engine-DTkVyDX-.js.map +1 -0
  42. package/dist/{eval-campaign-9MozgKL7.js → eval-campaign-BmptJj50.js} +2 -2
  43. package/dist/{eval-campaign-9MozgKL7.js.map → eval-campaign-BmptJj50.js.map} +1 -1
  44. package/dist/{exact-types-Dpw2LeHA.d.ts → exact-types-MaaFcllV.d.ts} +2 -2
  45. package/dist/{exact-types-Dpw2LeHA.d.ts.map → exact-types-MaaFcllV.d.ts.map} +1 -1
  46. package/dist/external-optimizer-contracts-BrxY2Sli.d.ts +32 -0
  47. package/dist/external-optimizer-contracts-BrxY2Sli.d.ts.map +1 -0
  48. package/dist/{extract-usage-CS391dOE.js → extract-usage-DZs601Va.js} +2 -2
  49. package/dist/{extract-usage-CS391dOE.js.map → extract-usage-DZs601Va.js.map} +1 -1
  50. package/dist/{feedback-trajectory-CoNep7rl.d.ts → feedback-trajectory-BJUWOkJM.d.ts} +3 -3
  51. package/dist/{feedback-trajectory-CoNep7rl.d.ts.map → feedback-trajectory-BJUWOkJM.d.ts.map} +1 -1
  52. package/dist/fuzz.d.ts +1 -1
  53. package/dist/fuzz.js +1 -1
  54. package/dist/{hf-dataset-DBJXXoY1.js → hf-dataset-XggBupCr.js} +2 -2
  55. package/dist/{hf-dataset-DBJXXoY1.js.map → hf-dataset-XggBupCr.js.map} +1 -1
  56. package/dist/hosted/index.d.ts +3 -3
  57. package/dist/{index-D0cxAdaV.d.ts → index-BTm_P9aC.d.ts} +11 -11
  58. package/dist/{index-D0cxAdaV.d.ts.map → index-BTm_P9aC.d.ts.map} +1 -1
  59. package/dist/{index-B2-IxCMB.d.ts → index-CWOPCJiw.d.ts} +2 -2
  60. package/dist/{index-B2-IxCMB.d.ts.map → index-CWOPCJiw.d.ts.map} +1 -1
  61. package/dist/{index-sMN_hI4E.d.ts → index-CtR1xh4V.d.ts} +3 -3
  62. package/dist/{index-sMN_hI4E.d.ts.map → index-CtR1xh4V.d.ts.map} +1 -1
  63. package/dist/{index-CjVYlVBK.d.ts → index-_66rVpwN.d.ts} +5 -5
  64. package/dist/{index-CjVYlVBK.d.ts.map → index-_66rVpwN.d.ts.map} +1 -1
  65. package/dist/index.d.ts +35 -56
  66. package/dist/index.d.ts.map +1 -1
  67. package/dist/index.js +51 -176
  68. package/dist/index.js.map +1 -1
  69. package/dist/{insight-report-CXd8VBDR.d.ts → insight-report-Bu5Wi9tG.d.ts} +4 -4
  70. package/dist/{insight-report-CXd8VBDR.d.ts.map → insight-report-Bu5Wi9tG.d.ts.map} +1 -1
  71. package/dist/{integrity-B-MLFz0I.d.ts → integrity-COTh3DTH.d.ts} +2 -2
  72. package/dist/{integrity-B-MLFz0I.d.ts.map → integrity-COTh3DTH.d.ts.map} +1 -1
  73. package/dist/kind-factory-CFxA0JQX.js +2133 -0
  74. package/dist/kind-factory-CFxA0JQX.js.map +1 -0
  75. package/dist/ledger-core/index.js +2 -1
  76. package/dist/{ledger-core-C0Yx1I14.js → ledger-core-Dxz0Rkwa.js} +3 -85
  77. package/dist/ledger-core-Dxz0Rkwa.js.map +1 -0
  78. package/dist/{llm-client-Cj3c7PEm.js → llm-client-bkztEfIx.js} +2 -2
  79. package/dist/{llm-client-Cj3c7PEm.js.map → llm-client-bkztEfIx.js.map} +1 -1
  80. package/dist/meta-eval/index.d.ts +2 -2
  81. package/dist/multishot/index.d.ts +2 -2
  82. package/dist/openapi.json +1 -1
  83. package/dist/{release-report-CoyvyLBs.d.ts → release-report-fZarvIm-.d.ts} +3 -3
  84. package/dist/{release-report-CoyvyLBs.d.ts.map → release-report-fZarvIm-.d.ts.map} +1 -1
  85. package/dist/{replay-DbIYwso6.d.ts → replay-DjG4IG60.d.ts} +34 -143
  86. package/dist/replay-DjG4IG60.d.ts.map +1 -0
  87. package/dist/{replay-Cb-4Vf0k.js → replay-SA4OB7O7.js} +48 -137
  88. package/dist/replay-SA4OB7O7.js.map +1 -0
  89. package/dist/reporting.d.ts +4 -4
  90. package/dist/{researcher-BCeOEjtR.d.ts → researcher-BxhtGfKa.d.ts} +5 -5
  91. package/dist/{researcher-BCeOEjtR.d.ts.map → researcher-BxhtGfKa.d.ts.map} +1 -1
  92. package/dist/{reward-hacking-sE2l_NV6.d.ts → reward-hacking-CqSLiV51.d.ts} +2 -2
  93. package/dist/{reward-hacking-sE2l_NV6.d.ts.map → reward-hacking-CqSLiV51.d.ts.map} +1 -1
  94. package/dist/rl.d.ts +5 -5
  95. package/dist/rl.js +1 -1
  96. package/dist/rollout/index.d.ts +1 -1
  97. package/dist/rollout/index.js +2 -2
  98. package/dist/{rollout-DQFl0UXA.js → rollout-8nj3mYvx.js} +2 -2
  99. package/dist/{rollout-DQFl0UXA.js.map → rollout-8nj3mYvx.js.map} +1 -1
  100. package/dist/{rubric-predictive-validity-w2klGv1u.d.ts → rubric-predictive-validity-DQBQj6uV.d.ts} +2 -2
  101. package/dist/{rubric-predictive-validity-w2klGv1u.d.ts.map → rubric-predictive-validity-DQBQj6uV.d.ts.map} +1 -1
  102. package/dist/{run-evidence-CbE0A8Xg.d.ts → run-evidence-C4RcRQT5.d.ts} +3 -3
  103. package/dist/{run-evidence-CbE0A8Xg.d.ts.map → run-evidence-C4RcRQT5.d.ts.map} +1 -1
  104. package/dist/{run-record-DwHMk1Ai.d.ts → run-record-CztDMXVF.d.ts} +2 -2
  105. package/dist/{run-record-DwHMk1Ai.d.ts.map → run-record-CztDMXVF.d.ts.map} +1 -1
  106. package/dist/{semantic-concept-judge-DYXDPZW0.js → semantic-concept-judge-BuIJ9IfB.js} +43 -6
  107. package/dist/semantic-concept-judge-BuIJ9IfB.js.map +1 -0
  108. package/dist/{server-DLEvyW2z.js → server-DaCpLfi0.js} +3 -3
  109. package/dist/{server-DLEvyW2z.js.map → server-DaCpLfi0.js.map} +1 -1
  110. package/dist/single-run-lock-BTTtPZ9N.js +989 -0
  111. package/dist/single-run-lock-BTTtPZ9N.js.map +1 -0
  112. package/dist/{skill-usage-Bv3G4VkA.d.ts → skill-usage-B-BFS8M2.d.ts} +54 -39
  113. package/dist/skill-usage-B-BFS8M2.d.ts.map +1 -0
  114. package/dist/{skillopt-optimization-method-CjKMZy0d.js → skillopt-optimization-method-BbGnCC53.js} +18 -802
  115. package/dist/skillopt-optimization-method-BbGnCC53.js.map +1 -0
  116. package/dist/{skillopt-optimization-method-CzfnA8O-.d.ts → skillopt-optimization-method-_s0Tub7Y.d.ts} +11 -39
  117. package/dist/skillopt-optimization-method-_s0Tub7Y.d.ts.map +1 -0
  118. package/dist/{statistics-mf70aXKp.d.ts → statistics-B5d0Zd-z.d.ts} +2 -2
  119. package/dist/{statistics-mf70aXKp.d.ts.map → statistics-B5d0Zd-z.d.ts.map} +1 -1
  120. package/dist/store-otlp-DX4fGIcf.js +757 -0
  121. package/dist/store-otlp-DX4fGIcf.js.map +1 -0
  122. package/dist/{summary-report-BKinV4yD.d.ts → summary-report-Cg7BifAM.d.ts} +3 -3
  123. package/dist/{summary-report-BKinV4yD.d.ts.map → summary-report-Cg7BifAM.d.ts.map} +1 -1
  124. package/dist/tool-groups-CdYq22lX.d.ts +258 -0
  125. package/dist/tool-groups-CdYq22lX.d.ts.map +1 -0
  126. package/dist/traces.d.ts +7 -6
  127. package/dist/traces.js +5 -5
  128. package/dist/{types-zFYez3PK.d.ts → types-BBFNHxSK.d.ts} +5 -5
  129. package/dist/{types-zFYez3PK.d.ts.map → types-BBFNHxSK.d.ts.map} +1 -1
  130. package/dist/{types-BtJhn8v6.d.ts → types-DoEYskCd.d.ts} +5 -5
  131. package/dist/{types-BtJhn8v6.d.ts.map → types-DoEYskCd.d.ts.map} +1 -1
  132. package/dist/{types-5q2T25iW.d.ts → types-uPrS6mD-.d.ts} +2 -2
  133. package/dist/{types-5q2T25iW.d.ts.map → types-uPrS6mD-.d.ts.map} +1 -1
  134. package/dist/usage-receipt-CgxMEBZq.js +134 -0
  135. package/dist/usage-receipt-CgxMEBZq.js.map +1 -0
  136. package/dist/wire/index.d.ts +3 -3
  137. package/dist/wire/index.js +1 -1
  138. package/docs/trace-analysis.md +170 -484
  139. package/package.json +1 -2
  140. package/dist/analyze-runs-CPYxfPWT.d.ts +0 -72
  141. package/dist/analyze-runs-CPYxfPWT.d.ts.map +0 -1
  142. package/dist/benchmark-command-CMqVqReF.js.map +0 -1
  143. package/dist/campaign-BIBS-NHV.js.map +0 -1
  144. package/dist/completion-verifier-B4-IMYcS.d.ts.map +0 -1
  145. package/dist/cost-ledger-B1D3COAc.d.ts.map +0 -1
  146. package/dist/cost-ledger-CHDLA0Ss.js.map +0 -1
  147. package/dist/default-registry-PUhIVRWz.d.ts +0 -215
  148. package/dist/default-registry-PUhIVRWz.d.ts.map +0 -1
  149. package/dist/default-registry-lp5R0lve.js.map +0 -1
  150. package/dist/ledger-core-C0Yx1I14.js.map +0 -1
  151. package/dist/registry-C4yJTza7.d.ts +0 -178
  152. package/dist/registry-C4yJTza7.d.ts.map +0 -1
  153. package/dist/replay-Cb-4Vf0k.js.map +0 -1
  154. package/dist/replay-DbIYwso6.d.ts.map +0 -1
  155. package/dist/semantic-concept-judge-DYXDPZW0.js.map +0 -1
  156. package/dist/single-run-lock-D_bS5xhj.js +0 -318
  157. package/dist/single-run-lock-D_bS5xhj.js.map +0 -1
  158. package/dist/skill-usage-Bv3G4VkA.d.ts.map +0 -1
  159. package/dist/skillopt-optimization-method-CjKMZy0d.js.map +0 -1
  160. package/dist/skillopt-optimization-method-CzfnA8O-.d.ts.map +0 -1
  161. package/dist/store-otlp-BenKynPE.js +0 -1688
  162. package/dist/store-otlp-BenKynPE.js.map +0 -1
  163. package/dist/tools-DZGdROtG.js +0 -255
  164. package/dist/tools-DZGdROtG.js.map +0 -1
@@ -1,11 +1,16 @@
1
1
  import { c as ValidationError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
2
- import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-CHDLA0Ss.js";
3
- import { c as callLlmJson, f as maximumChargeForLlmRequest, l as costReceiptFromLlm, r as LlmResponseError, t as LlmCallError, u as costReceiptFromLlmError } from "./llm-client-Cj3c7PEm.js";
4
- import { d as TRACE_ANALYSIS_LIMITS, i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-BenKynPE.js";
5
- import { d as usageReceiptFromCostLedger, m as makeFinding, n as createRunCostLedger, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-D_bS5xhj.js";
6
- import { _ as canonicalString, c as writeLedgerFileAtomically, s as withLedgerFileLock, v as hashCanonical } from "./ledger-core-C0Yx1I14.js";
2
+ import { s as resolveModelPricing } from "./metrics-C9YY1OcL.js";
3
+ import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-CZ9diLxY.js";
4
+ import { c as callLlmJson, f as maximumChargeForLlmRequest, l as costReceiptFromLlm, r as LlmResponseError, t as LlmCallError, u as costReceiptFromLlmError } from "./llm-client-bkztEfIx.js";
5
+ import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-CFxA0JQX.js";
6
+ import { o as makeFinding, r as usageReceiptFromCostLedger } from "./usage-receipt-CgxMEBZq.js";
7
+ import { i as hashCanonical, r as canonicalString } from "./canonical-D011XM8r.js";
8
+ import { n as createRunCostLedger, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-BTTtPZ9N.js";
9
+ import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-DTkVyDX-.js";
10
+ import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-Dxz0Rkwa.js";
7
11
  import { E as pairedBootstrap } from "./statistics-ByxzSiOM.js";
8
- import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-D8dkki-J.js";
12
+ import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-DX4fGIcf.js";
13
+ import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-CYtcIF2V.js";
9
14
  import { z } from "zod";
10
15
  import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
11
16
  import * as nodePath from "node:path";
@@ -1161,17 +1166,26 @@ function isSha256(value) {
1161
1166
  //#region src/analyst/benchmark-implementation.ts
1162
1167
  const ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM = "sha256-canonical-source-manifest";
1163
1168
  const ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM = "sha256-canonical-file-manifest";
1164
- const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze(["package.json", "pnpm-lock.yaml"]);
1165
- const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "d06e3e3f171cc6bd5cf36b14325f507f34b1c6a9d5129ce8a85bf3c1681f8223";
1166
- /** The published benchmark evidence was produced at this package version.
1167
- * A release changes package.json's version field, which is part of the lock
1168
- * manifest but cannot change benchmark behavior, so the evidence stays bound
1169
- * to the digest at its creation. A test proves the current lock differs from
1170
- * the evidence lock by the version stamp alone; any real dependency change
1171
- * still forces a new benchmark run or explicit retirement of the evidence. */
1169
+ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
1170
+ "clients/python/pyproject.toml",
1171
+ "clients/python/uv.lock",
1172
+ "package.json",
1173
+ "pnpm-lock.yaml"
1174
+ ]);
1175
+ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "6f3dc59755ff2a26fba4c2c1ad670f436cdd183faf252a5b9c2d50e6b190b460";
1176
+ /** The published benchmark evidence was produced at this package version, by
1177
+ * the retired one-shot direct runner, before trace analysts moved to the
1178
+ * recursive DSPy RLM engine. Both evidence digests below are historical facts
1179
+ * about that artifact: the current implementation and dependency manifest have
1180
+ * since changed, so they cannot describe the current engine. A fresh certified
1181
+ * run must replace the published evidence before any accuracy number is
1182
+ * attributed to the engine that ships today. */
1172
1183
  const ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION = "0.137.0";
1173
1184
  const ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 = "1e03f2daed356d60316aabefb407ec1e437ac94d408d61eea4ae096e9c6fbb5b";
1185
+ const ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 = "4dba263b6256a30d56c7fdb2d992d3a953c0035d731f359b704db806f68f75ac";
1174
1186
  const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1187
+ "clients/python/src/agent_eval_rpc/dspy_rlm_bridge.py",
1188
+ "clients/python/src/agent_eval_rpc/optimizer_bridge_common.py",
1175
1189
  "src/analyst/benchmark-agentrx-calibration.ts",
1176
1190
  "src/analyst/benchmark-command-artifact.ts",
1177
1191
  "src/analyst/benchmark-command-persistence.ts",
@@ -1189,6 +1203,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1189
1203
  "src/analyst/benchmark-public-data.ts",
1190
1204
  "src/analyst/benchmark-public-errors.ts",
1191
1205
  "src/analyst/benchmark-public-model.ts",
1206
+ "src/analyst/benchmark-public-rlm.ts",
1192
1207
  "src/analyst/benchmark-public-types.ts",
1193
1208
  "src/analyst/benchmark-real-model.ts",
1194
1209
  "src/analyst/benchmark-report.ts",
@@ -1198,8 +1213,22 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1198
1213
  "src/analyst/benchmark-verification-artifacts.ts",
1199
1214
  "src/analyst/benchmark-verification-outcome.ts",
1200
1215
  "src/analyst/benchmark.ts",
1216
+ "src/analyst/dspy-rlm-engine.ts",
1217
+ "src/analyst/engine.ts",
1218
+ "src/analyst/exact-types.ts",
1219
+ "src/analyst/finding-signature.ts",
1220
+ "src/analyst/finding-subject.ts",
1221
+ "src/analyst/kind-factory.ts",
1222
+ "src/analyst/parse-tolerant.ts",
1223
+ "src/analyst/tool-groups.ts",
1224
+ "src/analyst/trace-tool-callback.ts",
1201
1225
  "src/analyst/types.ts",
1202
1226
  "src/analyst/usage-receipt.ts",
1227
+ "src/campaign/external-optimizer-contracts.ts",
1228
+ "src/campaign/external-optimizer-http.ts",
1229
+ "src/campaign/external-optimizer-model-proxy.ts",
1230
+ "src/campaign/external-optimizer-resources.ts",
1231
+ "src/campaign/external-optimizer-subprocess.ts",
1203
1232
  "src/campaign/search-ledger-errors.ts",
1204
1233
  "src/campaign/search-ledger-file.ts",
1205
1234
  "src/campaign/single-run-lock.ts",
@@ -1210,6 +1239,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1210
1239
  "src/judge-calibration.ts",
1211
1240
  "src/ledger-core/atomic-file-lock.ts",
1212
1241
  "src/ledger-core/canonical.ts",
1242
+ "src/ledger-core/deep-freeze.ts",
1213
1243
  "src/ledger-core/index.ts",
1214
1244
  "src/ledger-core/journal-file.ts",
1215
1245
  "src/ledger-core/journal.ts",
@@ -1229,12 +1259,13 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
1229
1259
  "src/trace-analyst/store-otlp.ts",
1230
1260
  "src/trace-analyst/store-schemas.ts",
1231
1261
  "src/trace-analyst/store.ts",
1262
+ "src/trace-analyst/tools.ts",
1232
1263
  "src/trace-analyst/types.ts",
1233
1264
  "src/trace/attribute-vocabulary.ts",
1234
1265
  "src/trace/otlp-attributes.ts",
1235
1266
  "src/trace/raw-provider-sink.ts"
1236
1267
  ]);
1237
- const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "4dba263b6256a30d56c7fdb2d992d3a953c0035d731f359b704db806f68f75ac";
1268
+ const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "5e47fc9d9f49c468d3ef06c5d552925e6a026f9e22a75056a9c8c5a2879744c0";
1238
1269
  function analystBenchmarkImplementationDigest() {
1239
1270
  return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256;
1240
1271
  }
@@ -2485,7 +2516,11 @@ function publicBenchmarkError(error, secrets = []) {
2485
2516
  function redactSensitiveText(value, secrets) {
2486
2517
  let redacted = value;
2487
2518
  for (const secret of secrets) if (secret) redacted = redacted.replaceAll(secret, "[REDACTED]");
2488
- return redacted.replace(/\bBearer\s+[^\s"',;]+/gi, "Bearer [REDACTED]").replace(/\b(api[_-]?key|access[_-]?token|refresh[_-]?token|password|secret)\b\s*[:=]\s*[^\s"',;]+/gi, "$1=[REDACTED]").slice(0, 500);
2519
+ redacted = redacted.replace(/\bBearer\s+[^\s"',;]+/gi, "Bearer [REDACTED]").replace(/\b(api[_-]?key|access[_-]?token|refresh[_-]?token|password|secret)\b\s*[:=]\s*[^\s"',;]+/gi, "$1=[REDACTED]");
2520
+ if (redacted.length <= 500) return redacted;
2521
+ const head = redacted.slice(0, 180);
2522
+ const marker = `...[${redacted.length - 460} chars omitted]...`;
2523
+ return `${head}${marker}${redacted.slice(-(500 - head.length - marker.length))}`;
2489
2524
  }
2490
2525
  //#endregion
2491
2526
  //#region src/analyst/benchmark-response-cache.ts
@@ -2629,7 +2664,8 @@ const TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS = [
2629
2664
  128,
2630
2665
  64
2631
2666
  ];
2632
- function createPublicBenchmarkModelRunner(dataset, config) {
2667
+ /** One-shot JSON baseline. This is not a recursive trace analyst. */
2668
+ function createPublicBenchmarkDirectRunner(dataset, config) {
2633
2669
  const model = requiredString(config.model, "model");
2634
2670
  const baseUrl = requiredString(config.baseUrl, "baseUrl");
2635
2671
  const apiKey = requiredString(config.apiKey, "apiKey");
@@ -2652,9 +2688,9 @@ function createPublicBenchmarkModelRunner(dataset, config) {
2652
2688
  ...config.fetchImpl ? { fetch: config.fetchImpl } : {}
2653
2689
  };
2654
2690
  return {
2655
- id: "model",
2691
+ id: "direct",
2656
2692
  async analyze(input, context) {
2657
- const trajectoryId = trajectoryIdFromCaseId(dataset, context.caseId);
2693
+ const trajectoryId = trajectoryIdFromCaseId$1(dataset, context.caseId);
2658
2694
  const costTags = {
2659
2695
  analystId: actor,
2660
2696
  benchmarkCaseId: context.caseId,
@@ -2665,7 +2701,7 @@ function createPublicBenchmarkModelRunner(dataset, config) {
2665
2701
  let providerModel = model;
2666
2702
  let producedAt;
2667
2703
  let modelMetadata = {
2668
- analysisMode: "single-pass",
2704
+ analysisMode: "direct-baseline",
2669
2705
  outputAdapter,
2670
2706
  protocolSha256: publicBenchmarkProtocolSha256(dataset)
2671
2707
  };
@@ -2797,7 +2833,7 @@ function createPublicBenchmarkModelRunner(dataset, config) {
2797
2833
  trajectoryId,
2798
2834
  predictions: rawPredictions,
2799
2835
  store: input.traceStore,
2800
- analystId: "model",
2836
+ analystId: "direct",
2801
2837
  providerModel,
2802
2838
  producedAt: requiredString(producedAt ?? "", "finding producedAt"),
2803
2839
  ...context.signal ? { signal: context.signal } : {}
@@ -2818,7 +2854,7 @@ function createPublicBenchmarkModelRunner(dataset, config) {
2818
2854
  };
2819
2855
  } catch (error) {
2820
2856
  if (context.signal?.aborted) throw error;
2821
- if (isPaidCallControlError(error)) throw error;
2857
+ if (isPaidCallControlError$1(error)) throw error;
2822
2858
  return {
2823
2859
  findings: [],
2824
2860
  usage: usageReceiptFromCostLedger(costLedger, {
@@ -2856,7 +2892,7 @@ function assertCacheReceiptMatches(cached, receipt) {
2856
2892
  receipt
2857
2893
  });
2858
2894
  }
2859
- function isPaidCallControlError(error) {
2895
+ function isPaidCallControlError$1(error) {
2860
2896
  return error instanceof CostAccountingIncompleteError || error instanceof CostCallConflictError || error instanceof CostCeilingReachedError || error instanceof CostLedgerPersistenceError || error instanceof CostReceiptCaptureError || error instanceof CostReservationExceededError;
2861
2897
  }
2862
2898
  function receiptForProviderFailure(error, completedResult, model) {
@@ -2973,7 +3009,7 @@ async function publicBenchmarkPredictionsToFindings(options) {
2973
3009
  evidence_refs: [evidenceByStep.get(step)],
2974
3010
  recommended_action: prediction.recommended_action,
2975
3011
  metadata: {
2976
- analysis_mode: "single-pass",
3012
+ analysis_mode: "direct-baseline",
2977
3013
  model: options.providerModel
2978
3014
  },
2979
3015
  produced_at: options.producedAt,
@@ -3067,11 +3103,174 @@ Each finding's step MUST be the positive integer n from an existing assistant LL
3067
3103
  Never select an EVALUATOR, TOOL, CHAIN, final-verification, benchmark-verification, or message-<n> span.
3068
3104
  Before emitting a finding, inspect its candidate span's attributes.content and describe only the action shown there.
3069
3105
  When the trajectory has no incorrect steps, return an empty findings array.`;
3106
+ function trajectoryIdFromCaseId$1(dataset, caseId) {
3107
+ const prefix = dataset === "agentrx" ? "agentrx:" : "codetrace:";
3108
+ if (!caseId.startsWith(prefix) || caseId.length === prefix.length) throw new Error(`unexpected ${dataset} benchmark case id '${caseId}'`);
3109
+ return caseId.slice(prefix.length);
3110
+ }
3111
+ //#endregion
3112
+ //#region src/analyst/benchmark-public-rlm.ts
3113
+ const AGENT_RX_RLM_INSTRUCTIONS = `Analyze exactly one failed agent trajectory.
3114
+ Find the first unrecoverable critical failure, not every later symptom.
3115
+ Use the trace tools to inspect the action and its following observation.
3116
+ Emit zero findings only when evidence does not support a root cause.
3117
+ Otherwise emit exactly one finding whose subject is exactly one of:
3118
+ instruction-plan-adherence-failure
3119
+ invention-of-new-information
3120
+ invalid-invocation
3121
+ misinterpretation-of-tool-output-handoff-failure
3122
+ intent-plan-misalignment
3123
+ underspecified-user-intent
3124
+ intent-not-supported
3125
+ guardrails-triggered
3126
+ system-failure
3127
+ inconclusive
3128
+ Cite exactly one assistant span named step-<n> as trace://<trace-id>/span/step-<n>.
3129
+ The excerpt must quote the assistant action exactly.`;
3130
+ const CODE_TRACE_RLM_INSTRUCTIONS = `${CODE_TRACE_BENCH_ANALYST_PROMPT}
3131
+ Use the trace tools rather than asking for the whole trajectory in the prompt.
3132
+ Keep retrieved trace objects in Python variables.
3133
+ Never print an entire trace, full source file, or more than 12000 characters in one iteration.
3134
+ Build a compact table of assistant step ids, actions, following observations, and final verification.
3135
+ Inspect suspicious steps with viewSpans or searchSpan instead of repeatedly printing the table.
3136
+ Submit as soon as every state-changing assistant step has a supported verdict.
3137
+ For each incorrect step, set subject to incorrect-step-<n>.
3138
+ Cite the assistant span as trace://<URL-encoded-trace-id>/span/step-<n>.
3139
+ The evidence excerpt must be an exact quote from that span's action content.
3140
+ Return no finding for a clean trajectory.`;
3141
+ /** Public benchmark candidate that runs the actual recursive trace analyst. */
3142
+ function createPublicBenchmarkRlmRunner(dataset, config) {
3143
+ const costLedger = config.costLedger ?? new CostLedger();
3144
+ const limits = {
3145
+ maxIterations: config.dspyRlm?.maxIterations ?? 8,
3146
+ maxLlmCalls: config.dspyRlm?.maxLlmCalls ?? 4,
3147
+ maxToolCalls: config.dspyRlm?.maxToolCalls ?? 32,
3148
+ maxOutputChars: config.dspyRlm?.maxOutputChars ?? 8e3
3149
+ };
3150
+ const pricing = config.pricing ?? pricingForModel(config.model);
3151
+ const engine = createDspyRlmTraceEngine({
3152
+ baseUrl: config.baseUrl,
3153
+ apiKey: config.apiKey,
3154
+ model: config.model,
3155
+ maxOutputTokens: config.maxOutputTokens,
3156
+ timeoutMs: config.timeoutMs,
3157
+ maxCostUsd: config.maxCostUsdPerAnalysis ?? 1,
3158
+ pricing,
3159
+ ...config.dspyRlm?.runner ? { runner: config.dspyRlm.runner } : {}
3160
+ });
3161
+ const definition = publicBenchmarkDefinition(dataset, limits);
3162
+ return {
3163
+ id: "dspy-rlm",
3164
+ async analyze(input, context) {
3165
+ const trajectoryId = trajectoryIdFromCaseId(dataset, context.caseId);
3166
+ const tags = {
3167
+ benchmarkCaseId: context.caseId,
3168
+ benchmarkRepetition: String(context.repetition)
3169
+ };
3170
+ let usage;
3171
+ let rawFindings = [];
3172
+ try {
3173
+ if (!input.traceStore) throw new Error(`${dataset} DSPy RLM runner requires a trace store`);
3174
+ const completed = await runTraceAnalyst({
3175
+ definition,
3176
+ engine,
3177
+ store: input.traceStore,
3178
+ context: {
3179
+ runId: context.caseId,
3180
+ correlationId: `${context.caseId}:${context.repetition}`,
3181
+ costLedger,
3182
+ costPhase: "analyst.public-benchmark.dspy-rlm",
3183
+ tags,
3184
+ recordUsage: (receipt) => {
3185
+ usage = receipt;
3186
+ },
3187
+ signal: context.signal
3188
+ }
3189
+ });
3190
+ const producedAt = (/* @__PURE__ */ new Date()).toISOString();
3191
+ rawFindings = completed.findings.map((finding) => makeFinding({
3192
+ analyst_id: "dspy-rlm",
3193
+ area: dataset === "agentrx" ? "root-cause" : "incorrect",
3194
+ subject: finding.subject,
3195
+ claim: finding.claim,
3196
+ rationale: finding.rationale,
3197
+ severity: finding.severity,
3198
+ confidence: finding.confidence,
3199
+ evidence_refs: evidenceRefsFromRawFinding(finding),
3200
+ recommended_action: finding.recommended_action,
3201
+ metadata: {
3202
+ analysis_mode: "recursive",
3203
+ engine: "dspy-rlm",
3204
+ model: config.model
3205
+ },
3206
+ produced_at: producedAt
3207
+ }));
3208
+ const findings = adaptPublicBenchmarkFindings(dataset, trajectoryId, rawFindings, "dspy-rlm");
3209
+ if (dataset === "codetracebench") await validateCodeTraceFindingEvidence({
3210
+ trajectoryId,
3211
+ findings,
3212
+ store: input.traceStore,
3213
+ ...context.signal ? { signal: context.signal } : {}
3214
+ });
3215
+ return {
3216
+ findings,
3217
+ usage,
3218
+ metadata: {
3219
+ analysisMode: "recursive",
3220
+ engine: "dspy-rlm",
3221
+ protocolSha256: publicBenchmarkProtocolSha256(dataset),
3222
+ answer: completed.answer,
3223
+ trajectory: completed.trajectory,
3224
+ modelCalls: completed.modelCalls,
3225
+ toolCalls: completed.toolCalls,
3226
+ runtime: completed.runtime
3227
+ }
3228
+ };
3229
+ } catch (error) {
3230
+ if (context.signal?.aborted) throw error;
3231
+ if (isPaidCallControlError(error)) throw error;
3232
+ return {
3233
+ findings: [],
3234
+ usage,
3235
+ error: publicBenchmarkError(error, [config.apiKey]),
3236
+ metadata: {
3237
+ analysisMode: "recursive",
3238
+ engine: "dspy-rlm",
3239
+ rawFindings
3240
+ }
3241
+ };
3242
+ }
3243
+ }
3244
+ };
3245
+ }
3246
+ function publicBenchmarkDefinition(dataset, limits) {
3247
+ return {
3248
+ id: dataset === "agentrx" ? "agentrx-dspy-rlm" : "codetracebench-dspy-rlm",
3249
+ description: dataset === "agentrx" ? "Localizes the first unrecoverable root-cause step." : "Localizes every incorrect state-changing assistant step.",
3250
+ area: dataset === "agentrx" ? "root-cause" : "incorrect",
3251
+ version: "1.0.0",
3252
+ question: dataset === "agentrx" ? "What is the first unrecoverable root cause in this failed trajectory?" : "Which assistant steps are incorrect under the CodeTraceBench definition?",
3253
+ instructions: dataset === "agentrx" ? AGENT_RX_RLM_INSTRUCTIONS : CODE_TRACE_RLM_INSTRUCTIONS,
3254
+ toolGroup: "singleTrace",
3255
+ limits
3256
+ };
3257
+ }
3258
+ function pricingForModel(model) {
3259
+ const pricing = resolveModelPricing(model);
3260
+ if (!pricing) throw new Error(`no pricing is configured for '${model}'; provide PublicAnalystBenchmarkModelConfig.pricing`);
3261
+ return {
3262
+ inputUsdPerMillion: pricing.input * 1e3,
3263
+ outputUsdPerMillion: pricing.output * 1e3
3264
+ };
3265
+ }
3070
3266
  function trajectoryIdFromCaseId(dataset, caseId) {
3071
3267
  const prefix = dataset === "agentrx" ? "agentrx:" : "codetrace:";
3072
3268
  if (!caseId.startsWith(prefix) || caseId.length === prefix.length) throw new Error(`unexpected ${dataset} benchmark case id '${caseId}'`);
3073
3269
  return caseId.slice(prefix.length);
3074
3270
  }
3271
+ function isPaidCallControlError(error) {
3272
+ return error instanceof CostAccountingIncompleteError || error instanceof CostCallConflictError || error instanceof CostCeilingReachedError || error instanceof CostLedgerPersistenceError || error instanceof CostReceiptCaptureError || error instanceof CostReservationExceededError;
3273
+ }
3075
3274
  //#endregion
3076
3275
  //#region src/analyst/benchmark-command-persistence.ts
3077
3276
  const ANALYST_BENCHMARK_INITIALIZATION_COMPLETE_FILE = "initialization-complete.json";
@@ -3142,7 +3341,7 @@ function createRunIdentity(config, prepared) {
3142
3341
  analystProtocolSha256: publicBenchmarkProtocolSha256(config.dataset),
3143
3342
  implementationSha256: ANALYST_BENCHMARK_IMPLEMENTATION_SHA256,
3144
3343
  dependencyLockSha256: ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256,
3145
- runnerIds: ["empty", "model"]
3344
+ runnerIds: ["empty", "dspy-rlm"]
3146
3345
  },
3147
3346
  inputs: {
3148
3347
  labelsSha256: prepared.labelsSha256,
@@ -3347,7 +3546,7 @@ async function readProgress(path, runIdentitySha256, caseIds, repetitions) {
3347
3546
  previousRowSha256: parsed.previousRowSha256,
3348
3547
  observation
3349
3548
  }) !== parsed.rowSha256) throw new Error(`benchmark observation row ${index + 1} digest does not match its contents`);
3350
- if (!allowedCases.has(observation.caseId) || observation.runnerId !== "empty" && observation.runnerId !== "model" || observation.repetition >= repetitions || observation.executionIndex >= plannedObservationCount) throw new Error(`benchmark observation row ${index + 1} does not match a planned case, runner, and repetition`);
3549
+ if (!allowedCases.has(observation.caseId) || observation.runnerId !== "empty" && observation.runnerId !== "dspy-rlm" || observation.repetition >= repetitions || observation.executionIndex >= plannedObservationCount) throw new Error(`benchmark observation row ${index + 1} does not match a planned case, runner, and repetition`);
3351
3550
  if (executionIndexes.has(observation.executionIndex)) throw new Error(`duplicate benchmark executionIndex ${observation.executionIndex} at line ${index + 1}`);
3352
3551
  observations.push(observation);
3353
3552
  seen.add(key);
@@ -3802,7 +4001,7 @@ function assertCompletedArtifactMatchesRun(artifact, manifest, observations, pre
3802
4001
  if (canonicalJson(artifact.result.summaries) !== canonicalJson(expectedSummaries)) throw new Error("completed benchmark summaries do not match durable observations");
3803
4002
  const expectedComparisons = [compareAnalystRunners(artifact.result, {
3804
4003
  baselineRunnerId: "empty",
3805
- candidateRunnerId: "model",
4004
+ candidateRunnerId: "dspy-rlm",
3806
4005
  seed: config.seed
3807
4006
  })];
3808
4007
  if (canonicalJson(artifact.comparisons) !== canonicalJson(expectedComparisons)) throw new Error("completed benchmark comparisons do not match durable observations");
@@ -3951,8 +4150,8 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
3951
4150
  return benchmarkExitCode(artifact.result);
3952
4151
  }
3953
4152
  if (await regularFileExists(paths.report)) throw new Error(`benchmark report exists without a completed result; refusing ambiguous resume: ${paths.report}`);
3954
- const createModelRunner = dependencies.createModelRunner ?? ((dataset, model) => createPublicBenchmarkModelRunner(dataset, model));
3955
- const runners = [emptyPublicBenchmarkRunner(), createModelRunner(config.dataset, {
4153
+ const createAnalystRunner = dependencies.createAnalystRunner ?? ((dataset, model) => createPublicBenchmarkRlmRunner(dataset, model));
4154
+ const runners = [emptyPublicBenchmarkRunner(), createAnalystRunner(config.dataset, {
3956
4155
  ...config.model,
3957
4156
  costLedger,
3958
4157
  durability: {
@@ -4016,7 +4215,7 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
4016
4215
  assertSameObservations(result.observations, persisted.observations);
4017
4216
  const comparisons = [compareAnalystRunners(result, {
4018
4217
  baselineRunnerId: "empty",
4019
- candidateRunnerId: "model",
4218
+ candidateRunnerId: "dspy-rlm",
4020
4219
  seed: config.seed
4021
4220
  })];
4022
4221
  const codeTraceCalibration = config.dataset === "codetracebench" ? summarizeCodeTraceCalibration(result) : void 0;
@@ -4073,7 +4272,7 @@ const NON_SCORABLE_COST_ERRORS = /* @__PURE__ */ new Set([
4073
4272
  ]);
4074
4273
  function assertObservationAccountingComplete(observation, costLedger) {
4075
4274
  if (observation.error && NON_SCORABLE_COST_ERRORS.has(observation.error.class)) throw new CostAccountingIncompleteError(`Analyst benchmark stopped before scoring: ${observation.error.message}`);
4076
- if (observation.runnerId !== "model") return;
4275
+ if (observation.runnerId !== "dspy-rlm") return;
4077
4276
  const summary = costLedger.summary({
4078
4277
  channel: "analyst",
4079
4278
  tags: {
@@ -4081,7 +4280,7 @@ function assertObservationAccountingComplete(observation, costLedger) {
4081
4280
  benchmarkRepetition: String(observation.repetition)
4082
4281
  }
4083
4282
  });
4084
- if (!summary.accountingComplete || summary.pendingCalls > 0 || summary.unresolvedCalls > 0) throw accountingError(costLedger, "the model observation has incomplete cost accounting", {
4283
+ if (!summary.accountingComplete || summary.pendingCalls > 0 || summary.unresolvedCalls > 0) throw accountingError(costLedger, "the recursive analyst has incomplete cost accounting", {
4085
4284
  channel: "analyst",
4086
4285
  tags: {
4087
4286
  benchmarkCaseId: observation.caseId,
@@ -4099,7 +4298,7 @@ function accountingError(costLedger, reason, filter) {
4099
4298
  }
4100
4299
  const ANALYST_BENCHMARK_HELP = `agent-eval analyst-benchmark
4101
4300
 
4102
- Run a real-model trace analyst against public AgentRx or CodeTraceBench labels.
4301
+ Run the recursive DSPy trace analyst against public AgentRx or CodeTraceBench labels.
4103
4302
 
4104
4303
  Required:
4105
4304
  --dataset agentrx|codetracebench
@@ -4120,6 +4319,7 @@ Controls:
4120
4319
  --concurrency <positive integer> Parallel benchmark jobs. Default: 1
4121
4320
  --repetitions <positive integer> Runs per case and runner. Default: 1
4122
4321
  --max-output-tokens <positive> Model output limit per call. Default: 4096
4322
+ --python <executable> Python with agent-eval-rpc[dspy]. Default: python
4123
4323
  --timeout-ms <positive> Model analyst deadline per case. Default: 300000
4124
4324
  --max-cost-usd <positive> Run-wide spend limit. Default: 5
4125
4325
  --max-artifact-bytes <positive> Final evidence bytes per case. Default: ${DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES}
@@ -4142,6 +4342,8 @@ function parseCommandConfig(argv, env) {
4142
4342
  if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(apiKeyEnv)) throw new Error("--api-key-env must be a valid environment variable name");
4143
4343
  const apiKey = env[apiKeyEnv]?.trim();
4144
4344
  if (!apiKey) throw new Error(`--api-key-env points to an empty or missing variable: ${apiKeyEnv}`);
4345
+ const maxCostUsd = positiveFiniteFlag(flags, "max-cost-usd", 5);
4346
+ const python = flags.get("python")?.trim();
4145
4347
  return {
4146
4348
  dataset,
4147
4349
  labelsPath: requiredFlag(flags, "labels"),
@@ -4155,13 +4357,15 @@ function parseCommandConfig(argv, env) {
4155
4357
  apiKey,
4156
4358
  model: requiredFlag(flags, "model"),
4157
4359
  maxOutputTokens: positiveFlag(flags, "max-output-tokens", 4096),
4158
- timeoutMs: positiveFlag(flags, "timeout-ms", 3e5)
4360
+ timeoutMs: positiveFlag(flags, "timeout-ms", 3e5),
4361
+ maxCostUsdPerAnalysis: maxCostUsd,
4362
+ ...python ? { dspyRlm: { runner: { command: python } } } : {}
4159
4363
  },
4160
4364
  limit: positiveFlag(flags, "limit"),
4161
4365
  seed: integerFlag(flags, "seed", 0),
4162
4366
  concurrency: positiveFlag(flags, "concurrency", 1),
4163
4367
  repetitions: positiveFlag(flags, "repetitions", 1),
4164
- maxCostUsd: positiveFiniteFlag(flags, "max-cost-usd", 5),
4368
+ maxCostUsd,
4165
4369
  maxArtifactBytes: positiveFlag(flags, "max-artifact-bytes", DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES),
4166
4370
  apiKeyEnv,
4167
4371
  command: `agent-eval analyst-benchmark ${argv.filter((argument) => argument !== "--resume").map(shellQuote).join(" ")}`,
@@ -4206,6 +4410,7 @@ const KNOWN_FLAGS = /* @__PURE__ */ new Set([
4206
4410
  "concurrency",
4207
4411
  "repetitions",
4208
4412
  "max-output-tokens",
4413
+ "python",
4209
4414
  "timeout-ms",
4210
4415
  "max-cost-usd",
4211
4416
  "max-artifact-bytes"
@@ -4315,7 +4520,7 @@ function renderArtifactMarkdown(artifact) {
4315
4520
  return `${renderAnalystBenchmarkMarkdown(artifact.result, artifact.comparisons).trimEnd()}${calibrationMarkdown}${verificationMarkdown}\n\n${renderSelectionMarkdown(artifact.inputs.selection.report)}\n`;
4316
4521
  }
4317
4522
  function benchmarkExitCode(result) {
4318
- return result.summaries.find((summary) => summary.runnerId === "model")?.failedRuns ? 2 : 0;
4523
+ return result.summaries.find((summary) => summary.runnerId === "dspy-rlm")?.failedRuns ? 2 : 0;
4319
4524
  }
4320
4525
  function printSuccessSummary(artifact, paths) {
4321
4526
  const failures = artifact.result.summaries.reduce((total, summary) => total + summary.failedRuns, 0);
@@ -4327,6 +4532,6 @@ function shellQuote(value) {
4327
4532
  return /^[a-zA-Z0-9_./:@%+=,-]+$/.test(value) ? value : `'${value.replaceAll("'", "'\\''")}'`;
4328
4533
  }
4329
4534
  //#endregion
4330
- export { analystBenchmarkDependencyLockDigest as A, codeTracerPredictionsToFindings as B, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as C, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as D, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as E, ANALYST_BENCHMARK_OBSERVATIONS_FILE as F, normalizeBenchmarkLabel as G, agentRxPredictionsToFindings as H, AGENT_RX_UPSTREAM_REVISION as I, renderAgentRxCalibrationMarkdown as L, ANALYST_BENCHMARK_COST_LEDGER_FILE as M, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as N, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as O, ANALYST_BENCHMARK_MANIFEST_FILE as P, summarizeAgentRxCalibration as R, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as S, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as T, normalizeAgentRxCategory as U, agentRxBenchmarkCase as V, roundAgentRxStep as W, appendVerificationArtifactsToOtlp as _, renderCodeTraceCalibrationMarkdown as a, adaptPublicBenchmarkFindings as b, CODE_TRACE_BENCH_ANALYST_PROMPT as c, loadPublicBenchmarkRows as d, preparePublicAnalystBenchmark as f, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as g, selectPublicBenchmarkRows as h, readAnalystBenchmarkArtifact as i, analystBenchmarkImplementationDigest as j, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as k, createPublicBenchmarkModelRunner as l, publicBenchmarkSelectionReport as m, runAnalystBenchmarkCommand as n, summarizeCodeTraceCalibration as o, publicBenchmarkDistributions as p, renderAnalystBenchmarkMarkdown as r, compareAnalystRunners as s, ANALYST_BENCHMARK_HELP as t, publicBenchmarkProtocolSha256 as u, loadCodeTraceVerificationArtifacts as v, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as w, emptyPublicBenchmarkRunner as x, parseVerificationOutcome as y, codeTraceBenchCase as z };
4535
+ export { ANALYST_BENCHMARK_IMPLEMENTATION_FILES as A, summarizeAgentRxCalibration as B, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as C, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as D, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as E, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as F, normalizeAgentRxCategory as G, codeTracerPredictionsToFindings as H, ANALYST_BENCHMARK_MANIFEST_FILE as I, roundAgentRxStep as K, ANALYST_BENCHMARK_OBSERVATIONS_FILE as L, analystBenchmarkDependencyLockDigest as M, analystBenchmarkImplementationDigest as N, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as O, ANALYST_BENCHMARK_COST_LEDGER_FILE as P, AGENT_RX_UPSTREAM_REVISION as R, emptyPublicBenchmarkRunner as S, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as T, agentRxBenchmarkCase as U, codeTraceBenchCase as V, agentRxPredictionsToFindings as W, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as _, renderCodeTraceCalibrationMarkdown as a, parseVerificationOutcome as b, createPublicBenchmarkRlmRunner as c, publicBenchmarkProtocolSha256 as d, loadPublicBenchmarkRows as f, selectPublicBenchmarkRows as g, publicBenchmarkSelectionReport as h, readAnalystBenchmarkArtifact as i, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as j, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as k, CODE_TRACE_BENCH_ANALYST_PROMPT as l, publicBenchmarkDistributions as m, runAnalystBenchmarkCommand as n, summarizeCodeTraceCalibration as o, preparePublicAnalystBenchmark as p, normalizeBenchmarkLabel as q, renderAnalystBenchmarkMarkdown as r, compareAnalystRunners as s, ANALYST_BENCHMARK_HELP as t, createPublicBenchmarkDirectRunner as u, appendVerificationArtifactsToOtlp as v, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as w, adaptPublicBenchmarkFindings as x, loadCodeTraceVerificationArtifacts as y, renderAgentRxCalibrationMarkdown as z };
4331
4536
 
4332
- //# sourceMappingURL=benchmark-command-CMqVqReF.js.map
4537
+ //# sourceMappingURL=benchmark-command-BKfjOBJ5.js.map