@tangle-network/agent-eval 0.138.0 → 0.139.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -1
- package/dist/analyst/index.d.ts +41 -94
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +9 -24
- package/dist/analyst/index.js.map +1 -1
- package/dist/{benchmark-D8dkki-J.js → benchmark-CYtcIF2V.js} +2 -2
- package/dist/{benchmark-D8dkki-J.js.map → benchmark-CYtcIF2V.js.map} +1 -1
- package/dist/{benchmark-DlQgU_XI.d.ts → benchmark-DDVdWcwA.d.ts} +3 -3
- package/dist/{benchmark-DlQgU_XI.d.ts.map → benchmark-DDVdWcwA.d.ts.map} +1 -1
- package/dist/{benchmark-command-CMqVqReF.js → benchmark-command-BKfjOBJ5.js} +243 -38
- package/dist/benchmark-command-BKfjOBJ5.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BJ_xK5rQ.js → benchmarks-zxhy1QV3.js} +4 -4
- package/dist/{benchmarks-BJ_xK5rQ.js.map → benchmarks-zxhy1QV3.js.map} +1 -1
- package/dist/campaign/index.d.ts +5 -5
- package/dist/campaign/index.js +4 -3
- package/dist/{campaign-BIBS-NHV.js → campaign-DrS6_hLd.js} +10 -9
- package/dist/campaign-DrS6_hLd.js.map +1 -0
- package/dist/canonical-D011XM8r.js +86 -0
- package/dist/canonical-D011XM8r.js.map +1 -0
- package/dist/cli.js +3 -3
- package/dist/{client-BwPKohkJ.d.ts → client-BohnDFBq.d.ts} +4 -4
- package/dist/{client-BwPKohkJ.d.ts.map → client-BohnDFBq.d.ts.map} +1 -1
- package/dist/{completion-verifier-B4-IMYcS.d.ts → completion-verifier-IPoP4fQO.d.ts} +178 -4
- package/dist/completion-verifier-IPoP4fQO.d.ts.map +1 -0
- package/dist/contract/index.d.ts +10 -10
- package/dist/contract/index.js +8 -7
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/{cost-ledger-CHDLA0Ss.js → cost-ledger-CZ9diLxY.js} +7 -7
- package/dist/cost-ledger-CZ9diLxY.js.map +1 -0
- package/dist/{cost-ledger-B1D3COAc.d.ts → cost-ledger-DKgyIWRj.d.ts} +5 -2
- package/dist/cost-ledger-DKgyIWRj.d.ts.map +1 -0
- package/dist/default-registry-B8vf7Rmf.d.ts +118 -0
- package/dist/default-registry-B8vf7Rmf.d.ts.map +1 -0
- package/dist/{default-registry-lp5R0lve.js → default-registry-BgJJItGr.js} +57 -1532
- package/dist/default-registry-BgJJItGr.js.map +1 -0
- package/dist/dspy-rlm-engine-DTkVyDX-.js +344 -0
- package/dist/dspy-rlm-engine-DTkVyDX-.js.map +1 -0
- package/dist/{eval-campaign-9MozgKL7.js → eval-campaign-BmptJj50.js} +2 -2
- package/dist/{eval-campaign-9MozgKL7.js.map → eval-campaign-BmptJj50.js.map} +1 -1
- package/dist/{exact-types-Dpw2LeHA.d.ts → exact-types-MaaFcllV.d.ts} +2 -2
- package/dist/{exact-types-Dpw2LeHA.d.ts.map → exact-types-MaaFcllV.d.ts.map} +1 -1
- package/dist/external-optimizer-contracts-BrxY2Sli.d.ts +32 -0
- package/dist/external-optimizer-contracts-BrxY2Sli.d.ts.map +1 -0
- package/dist/{extract-usage-CS391dOE.js → extract-usage-DZs601Va.js} +2 -2
- package/dist/{extract-usage-CS391dOE.js.map → extract-usage-DZs601Va.js.map} +1 -1
- package/dist/{feedback-trajectory-CoNep7rl.d.ts → feedback-trajectory-BJUWOkJM.d.ts} +3 -3
- package/dist/{feedback-trajectory-CoNep7rl.d.ts.map → feedback-trajectory-BJUWOkJM.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/{hf-dataset-DBJXXoY1.js → hf-dataset-XggBupCr.js} +2 -2
- package/dist/{hf-dataset-DBJXXoY1.js.map → hf-dataset-XggBupCr.js.map} +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-D0cxAdaV.d.ts → index-BTm_P9aC.d.ts} +11 -11
- package/dist/{index-D0cxAdaV.d.ts.map → index-BTm_P9aC.d.ts.map} +1 -1
- package/dist/{index-B2-IxCMB.d.ts → index-CWOPCJiw.d.ts} +2 -2
- package/dist/{index-B2-IxCMB.d.ts.map → index-CWOPCJiw.d.ts.map} +1 -1
- package/dist/{index-sMN_hI4E.d.ts → index-CtR1xh4V.d.ts} +3 -3
- package/dist/{index-sMN_hI4E.d.ts.map → index-CtR1xh4V.d.ts.map} +1 -1
- package/dist/{index-CjVYlVBK.d.ts → index-_66rVpwN.d.ts} +5 -5
- package/dist/{index-CjVYlVBK.d.ts.map → index-_66rVpwN.d.ts.map} +1 -1
- package/dist/index.d.ts +35 -56
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +51 -176
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-CXd8VBDR.d.ts → insight-report-Bu5Wi9tG.d.ts} +4 -4
- package/dist/{insight-report-CXd8VBDR.d.ts.map → insight-report-Bu5Wi9tG.d.ts.map} +1 -1
- package/dist/{integrity-B-MLFz0I.d.ts → integrity-COTh3DTH.d.ts} +2 -2
- package/dist/{integrity-B-MLFz0I.d.ts.map → integrity-COTh3DTH.d.ts.map} +1 -1
- package/dist/kind-factory-CFxA0JQX.js +2133 -0
- package/dist/kind-factory-CFxA0JQX.js.map +1 -0
- package/dist/ledger-core/index.js +2 -1
- package/dist/{ledger-core-C0Yx1I14.js → ledger-core-Dxz0Rkwa.js} +3 -85
- package/dist/ledger-core-Dxz0Rkwa.js.map +1 -0
- package/dist/{llm-client-Cj3c7PEm.js → llm-client-bkztEfIx.js} +2 -2
- package/dist/{llm-client-Cj3c7PEm.js.map → llm-client-bkztEfIx.js.map} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{release-report-CoyvyLBs.d.ts → release-report-fZarvIm-.d.ts} +3 -3
- package/dist/{release-report-CoyvyLBs.d.ts.map → release-report-fZarvIm-.d.ts.map} +1 -1
- package/dist/{replay-DbIYwso6.d.ts → replay-DjG4IG60.d.ts} +34 -143
- package/dist/replay-DjG4IG60.d.ts.map +1 -0
- package/dist/{replay-Cb-4Vf0k.js → replay-SA4OB7O7.js} +48 -137
- package/dist/replay-SA4OB7O7.js.map +1 -0
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-BCeOEjtR.d.ts → researcher-BxhtGfKa.d.ts} +5 -5
- package/dist/{researcher-BCeOEjtR.d.ts.map → researcher-BxhtGfKa.d.ts.map} +1 -1
- package/dist/{reward-hacking-sE2l_NV6.d.ts → reward-hacking-CqSLiV51.d.ts} +2 -2
- package/dist/{reward-hacking-sE2l_NV6.d.ts.map → reward-hacking-CqSLiV51.d.ts.map} +1 -1
- package/dist/rl.d.ts +5 -5
- package/dist/rl.js +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-DQFl0UXA.js → rollout-8nj3mYvx.js} +2 -2
- package/dist/{rollout-DQFl0UXA.js.map → rollout-8nj3mYvx.js.map} +1 -1
- package/dist/{rubric-predictive-validity-w2klGv1u.d.ts → rubric-predictive-validity-DQBQj6uV.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-w2klGv1u.d.ts.map → rubric-predictive-validity-DQBQj6uV.d.ts.map} +1 -1
- package/dist/{run-evidence-CbE0A8Xg.d.ts → run-evidence-C4RcRQT5.d.ts} +3 -3
- package/dist/{run-evidence-CbE0A8Xg.d.ts.map → run-evidence-C4RcRQT5.d.ts.map} +1 -1
- package/dist/{run-record-DwHMk1Ai.d.ts → run-record-CztDMXVF.d.ts} +2 -2
- package/dist/{run-record-DwHMk1Ai.d.ts.map → run-record-CztDMXVF.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-DYXDPZW0.js → semantic-concept-judge-BuIJ9IfB.js} +43 -6
- package/dist/semantic-concept-judge-BuIJ9IfB.js.map +1 -0
- package/dist/{server-DLEvyW2z.js → server-DaCpLfi0.js} +3 -3
- package/dist/{server-DLEvyW2z.js.map → server-DaCpLfi0.js.map} +1 -1
- package/dist/single-run-lock-BTTtPZ9N.js +989 -0
- package/dist/single-run-lock-BTTtPZ9N.js.map +1 -0
- package/dist/{skill-usage-Bv3G4VkA.d.ts → skill-usage-B-BFS8M2.d.ts} +54 -39
- package/dist/skill-usage-B-BFS8M2.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-CjKMZy0d.js → skillopt-optimization-method-BbGnCC53.js} +18 -802
- package/dist/skillopt-optimization-method-BbGnCC53.js.map +1 -0
- package/dist/{skillopt-optimization-method-CzfnA8O-.d.ts → skillopt-optimization-method-_s0Tub7Y.d.ts} +11 -39
- package/dist/skillopt-optimization-method-_s0Tub7Y.d.ts.map +1 -0
- package/dist/{statistics-mf70aXKp.d.ts → statistics-B5d0Zd-z.d.ts} +2 -2
- package/dist/{statistics-mf70aXKp.d.ts.map → statistics-B5d0Zd-z.d.ts.map} +1 -1
- package/dist/store-otlp-DX4fGIcf.js +757 -0
- package/dist/store-otlp-DX4fGIcf.js.map +1 -0
- package/dist/{summary-report-BKinV4yD.d.ts → summary-report-Cg7BifAM.d.ts} +3 -3
- package/dist/{summary-report-BKinV4yD.d.ts.map → summary-report-Cg7BifAM.d.ts.map} +1 -1
- package/dist/tool-groups-CdYq22lX.d.ts +258 -0
- package/dist/tool-groups-CdYq22lX.d.ts.map +1 -0
- package/dist/traces.d.ts +7 -6
- package/dist/traces.js +5 -5
- package/dist/{types-zFYez3PK.d.ts → types-BBFNHxSK.d.ts} +5 -5
- package/dist/{types-zFYez3PK.d.ts.map → types-BBFNHxSK.d.ts.map} +1 -1
- package/dist/{types-BtJhn8v6.d.ts → types-DoEYskCd.d.ts} +5 -5
- package/dist/{types-BtJhn8v6.d.ts.map → types-DoEYskCd.d.ts.map} +1 -1
- package/dist/{types-5q2T25iW.d.ts → types-uPrS6mD-.d.ts} +2 -2
- package/dist/{types-5q2T25iW.d.ts.map → types-uPrS6mD-.d.ts.map} +1 -1
- package/dist/usage-receipt-CgxMEBZq.js +134 -0
- package/dist/usage-receipt-CgxMEBZq.js.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/trace-analysis.md +170 -484
- package/package.json +1 -2
- package/dist/analyze-runs-CPYxfPWT.d.ts +0 -72
- package/dist/analyze-runs-CPYxfPWT.d.ts.map +0 -1
- package/dist/benchmark-command-CMqVqReF.js.map +0 -1
- package/dist/campaign-BIBS-NHV.js.map +0 -1
- package/dist/completion-verifier-B4-IMYcS.d.ts.map +0 -1
- package/dist/cost-ledger-B1D3COAc.d.ts.map +0 -1
- package/dist/cost-ledger-CHDLA0Ss.js.map +0 -1
- package/dist/default-registry-PUhIVRWz.d.ts +0 -215
- package/dist/default-registry-PUhIVRWz.d.ts.map +0 -1
- package/dist/default-registry-lp5R0lve.js.map +0 -1
- package/dist/ledger-core-C0Yx1I14.js.map +0 -1
- package/dist/registry-C4yJTza7.d.ts +0 -178
- package/dist/registry-C4yJTza7.d.ts.map +0 -1
- package/dist/replay-Cb-4Vf0k.js.map +0 -1
- package/dist/replay-DbIYwso6.d.ts.map +0 -1
- package/dist/semantic-concept-judge-DYXDPZW0.js.map +0 -1
- package/dist/single-run-lock-D_bS5xhj.js +0 -318
- package/dist/single-run-lock-D_bS5xhj.js.map +0 -1
- package/dist/skill-usage-Bv3G4VkA.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CjKMZy0d.js.map +0 -1
- package/dist/skillopt-optimization-method-CzfnA8O-.d.ts.map +0 -1
- package/dist/store-otlp-BenKynPE.js +0 -1688
- package/dist/store-otlp-BenKynPE.js.map +0 -1
- package/dist/tools-DZGdROtG.js +0 -255
- package/dist/tools-DZGdROtG.js.map +0 -1
|
@@ -1,11 +1,16 @@
|
|
|
1
1
|
import { c as ValidationError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
|
|
2
|
-
import {
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
5
|
-
import {
|
|
6
|
-
import {
|
|
2
|
+
import { s as resolveModelPricing } from "./metrics-C9YY1OcL.js";
|
|
3
|
+
import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-CZ9diLxY.js";
|
|
4
|
+
import { c as callLlmJson, f as maximumChargeForLlmRequest, l as costReceiptFromLlm, r as LlmResponseError, t as LlmCallError, u as costReceiptFromLlmError } from "./llm-client-bkztEfIx.js";
|
|
5
|
+
import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-CFxA0JQX.js";
|
|
6
|
+
import { o as makeFinding, r as usageReceiptFromCostLedger } from "./usage-receipt-CgxMEBZq.js";
|
|
7
|
+
import { i as hashCanonical, r as canonicalString } from "./canonical-D011XM8r.js";
|
|
8
|
+
import { n as createRunCostLedger, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-BTTtPZ9N.js";
|
|
9
|
+
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-DTkVyDX-.js";
|
|
10
|
+
import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-Dxz0Rkwa.js";
|
|
7
11
|
import { E as pairedBootstrap } from "./statistics-ByxzSiOM.js";
|
|
8
|
-
import { i as
|
|
12
|
+
import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-DX4fGIcf.js";
|
|
13
|
+
import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-CYtcIF2V.js";
|
|
9
14
|
import { z } from "zod";
|
|
10
15
|
import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
|
|
11
16
|
import * as nodePath from "node:path";
|
|
@@ -1161,17 +1166,26 @@ function isSha256(value) {
|
|
|
1161
1166
|
//#region src/analyst/benchmark-implementation.ts
|
|
1162
1167
|
const ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM = "sha256-canonical-source-manifest";
|
|
1163
1168
|
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM = "sha256-canonical-file-manifest";
|
|
1164
|
-
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
|
|
1165
|
-
|
|
1166
|
-
|
|
1167
|
-
|
|
1168
|
-
|
|
1169
|
-
|
|
1170
|
-
|
|
1171
|
-
|
|
1169
|
+
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
|
|
1170
|
+
"clients/python/pyproject.toml",
|
|
1171
|
+
"clients/python/uv.lock",
|
|
1172
|
+
"package.json",
|
|
1173
|
+
"pnpm-lock.yaml"
|
|
1174
|
+
]);
|
|
1175
|
+
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "6f3dc59755ff2a26fba4c2c1ad670f436cdd183faf252a5b9c2d50e6b190b460";
|
|
1176
|
+
/** The published benchmark evidence was produced at this package version, by
|
|
1177
|
+
* the retired one-shot direct runner, before trace analysts moved to the
|
|
1178
|
+
* recursive DSPy RLM engine. Both evidence digests below are historical facts
|
|
1179
|
+
* about that artifact: the current implementation and dependency manifest have
|
|
1180
|
+
* since changed, so they cannot describe the current engine. A fresh certified
|
|
1181
|
+
* run must replace the published evidence before any accuracy number is
|
|
1182
|
+
* attributed to the engine that ships today. */
|
|
1172
1183
|
const ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION = "0.137.0";
|
|
1173
1184
|
const ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 = "1e03f2daed356d60316aabefb407ec1e437ac94d408d61eea4ae096e9c6fbb5b";
|
|
1185
|
+
const ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 = "4dba263b6256a30d56c7fdb2d992d3a953c0035d731f359b704db806f68f75ac";
|
|
1174
1186
|
const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
1187
|
+
"clients/python/src/agent_eval_rpc/dspy_rlm_bridge.py",
|
|
1188
|
+
"clients/python/src/agent_eval_rpc/optimizer_bridge_common.py",
|
|
1175
1189
|
"src/analyst/benchmark-agentrx-calibration.ts",
|
|
1176
1190
|
"src/analyst/benchmark-command-artifact.ts",
|
|
1177
1191
|
"src/analyst/benchmark-command-persistence.ts",
|
|
@@ -1189,6 +1203,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1189
1203
|
"src/analyst/benchmark-public-data.ts",
|
|
1190
1204
|
"src/analyst/benchmark-public-errors.ts",
|
|
1191
1205
|
"src/analyst/benchmark-public-model.ts",
|
|
1206
|
+
"src/analyst/benchmark-public-rlm.ts",
|
|
1192
1207
|
"src/analyst/benchmark-public-types.ts",
|
|
1193
1208
|
"src/analyst/benchmark-real-model.ts",
|
|
1194
1209
|
"src/analyst/benchmark-report.ts",
|
|
@@ -1198,8 +1213,22 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1198
1213
|
"src/analyst/benchmark-verification-artifacts.ts",
|
|
1199
1214
|
"src/analyst/benchmark-verification-outcome.ts",
|
|
1200
1215
|
"src/analyst/benchmark.ts",
|
|
1216
|
+
"src/analyst/dspy-rlm-engine.ts",
|
|
1217
|
+
"src/analyst/engine.ts",
|
|
1218
|
+
"src/analyst/exact-types.ts",
|
|
1219
|
+
"src/analyst/finding-signature.ts",
|
|
1220
|
+
"src/analyst/finding-subject.ts",
|
|
1221
|
+
"src/analyst/kind-factory.ts",
|
|
1222
|
+
"src/analyst/parse-tolerant.ts",
|
|
1223
|
+
"src/analyst/tool-groups.ts",
|
|
1224
|
+
"src/analyst/trace-tool-callback.ts",
|
|
1201
1225
|
"src/analyst/types.ts",
|
|
1202
1226
|
"src/analyst/usage-receipt.ts",
|
|
1227
|
+
"src/campaign/external-optimizer-contracts.ts",
|
|
1228
|
+
"src/campaign/external-optimizer-http.ts",
|
|
1229
|
+
"src/campaign/external-optimizer-model-proxy.ts",
|
|
1230
|
+
"src/campaign/external-optimizer-resources.ts",
|
|
1231
|
+
"src/campaign/external-optimizer-subprocess.ts",
|
|
1203
1232
|
"src/campaign/search-ledger-errors.ts",
|
|
1204
1233
|
"src/campaign/search-ledger-file.ts",
|
|
1205
1234
|
"src/campaign/single-run-lock.ts",
|
|
@@ -1210,6 +1239,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1210
1239
|
"src/judge-calibration.ts",
|
|
1211
1240
|
"src/ledger-core/atomic-file-lock.ts",
|
|
1212
1241
|
"src/ledger-core/canonical.ts",
|
|
1242
|
+
"src/ledger-core/deep-freeze.ts",
|
|
1213
1243
|
"src/ledger-core/index.ts",
|
|
1214
1244
|
"src/ledger-core/journal-file.ts",
|
|
1215
1245
|
"src/ledger-core/journal.ts",
|
|
@@ -1229,12 +1259,13 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1229
1259
|
"src/trace-analyst/store-otlp.ts",
|
|
1230
1260
|
"src/trace-analyst/store-schemas.ts",
|
|
1231
1261
|
"src/trace-analyst/store.ts",
|
|
1262
|
+
"src/trace-analyst/tools.ts",
|
|
1232
1263
|
"src/trace-analyst/types.ts",
|
|
1233
1264
|
"src/trace/attribute-vocabulary.ts",
|
|
1234
1265
|
"src/trace/otlp-attributes.ts",
|
|
1235
1266
|
"src/trace/raw-provider-sink.ts"
|
|
1236
1267
|
]);
|
|
1237
|
-
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
1268
|
+
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "5e47fc9d9f49c468d3ef06c5d552925e6a026f9e22a75056a9c8c5a2879744c0";
|
|
1238
1269
|
function analystBenchmarkImplementationDigest() {
|
|
1239
1270
|
return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256;
|
|
1240
1271
|
}
|
|
@@ -2485,7 +2516,11 @@ function publicBenchmarkError(error, secrets = []) {
|
|
|
2485
2516
|
function redactSensitiveText(value, secrets) {
|
|
2486
2517
|
let redacted = value;
|
|
2487
2518
|
for (const secret of secrets) if (secret) redacted = redacted.replaceAll(secret, "[REDACTED]");
|
|
2488
|
-
|
|
2519
|
+
redacted = redacted.replace(/\bBearer\s+[^\s"',;]+/gi, "Bearer [REDACTED]").replace(/\b(api[_-]?key|access[_-]?token|refresh[_-]?token|password|secret)\b\s*[:=]\s*[^\s"',;]+/gi, "$1=[REDACTED]");
|
|
2520
|
+
if (redacted.length <= 500) return redacted;
|
|
2521
|
+
const head = redacted.slice(0, 180);
|
|
2522
|
+
const marker = `...[${redacted.length - 460} chars omitted]...`;
|
|
2523
|
+
return `${head}${marker}${redacted.slice(-(500 - head.length - marker.length))}`;
|
|
2489
2524
|
}
|
|
2490
2525
|
//#endregion
|
|
2491
2526
|
//#region src/analyst/benchmark-response-cache.ts
|
|
@@ -2629,7 +2664,8 @@ const TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS = [
|
|
|
2629
2664
|
128,
|
|
2630
2665
|
64
|
|
2631
2666
|
];
|
|
2632
|
-
|
|
2667
|
+
/** One-shot JSON baseline. This is not a recursive trace analyst. */
|
|
2668
|
+
function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
2633
2669
|
const model = requiredString(config.model, "model");
|
|
2634
2670
|
const baseUrl = requiredString(config.baseUrl, "baseUrl");
|
|
2635
2671
|
const apiKey = requiredString(config.apiKey, "apiKey");
|
|
@@ -2652,9 +2688,9 @@ function createPublicBenchmarkModelRunner(dataset, config) {
|
|
|
2652
2688
|
...config.fetchImpl ? { fetch: config.fetchImpl } : {}
|
|
2653
2689
|
};
|
|
2654
2690
|
return {
|
|
2655
|
-
id: "
|
|
2691
|
+
id: "direct",
|
|
2656
2692
|
async analyze(input, context) {
|
|
2657
|
-
const trajectoryId = trajectoryIdFromCaseId(dataset, context.caseId);
|
|
2693
|
+
const trajectoryId = trajectoryIdFromCaseId$1(dataset, context.caseId);
|
|
2658
2694
|
const costTags = {
|
|
2659
2695
|
analystId: actor,
|
|
2660
2696
|
benchmarkCaseId: context.caseId,
|
|
@@ -2665,7 +2701,7 @@ function createPublicBenchmarkModelRunner(dataset, config) {
|
|
|
2665
2701
|
let providerModel = model;
|
|
2666
2702
|
let producedAt;
|
|
2667
2703
|
let modelMetadata = {
|
|
2668
|
-
analysisMode: "
|
|
2704
|
+
analysisMode: "direct-baseline",
|
|
2669
2705
|
outputAdapter,
|
|
2670
2706
|
protocolSha256: publicBenchmarkProtocolSha256(dataset)
|
|
2671
2707
|
};
|
|
@@ -2797,7 +2833,7 @@ function createPublicBenchmarkModelRunner(dataset, config) {
|
|
|
2797
2833
|
trajectoryId,
|
|
2798
2834
|
predictions: rawPredictions,
|
|
2799
2835
|
store: input.traceStore,
|
|
2800
|
-
analystId: "
|
|
2836
|
+
analystId: "direct",
|
|
2801
2837
|
providerModel,
|
|
2802
2838
|
producedAt: requiredString(producedAt ?? "", "finding producedAt"),
|
|
2803
2839
|
...context.signal ? { signal: context.signal } : {}
|
|
@@ -2818,7 +2854,7 @@ function createPublicBenchmarkModelRunner(dataset, config) {
|
|
|
2818
2854
|
};
|
|
2819
2855
|
} catch (error) {
|
|
2820
2856
|
if (context.signal?.aborted) throw error;
|
|
2821
|
-
if (isPaidCallControlError(error)) throw error;
|
|
2857
|
+
if (isPaidCallControlError$1(error)) throw error;
|
|
2822
2858
|
return {
|
|
2823
2859
|
findings: [],
|
|
2824
2860
|
usage: usageReceiptFromCostLedger(costLedger, {
|
|
@@ -2856,7 +2892,7 @@ function assertCacheReceiptMatches(cached, receipt) {
|
|
|
2856
2892
|
receipt
|
|
2857
2893
|
});
|
|
2858
2894
|
}
|
|
2859
|
-
function isPaidCallControlError(error) {
|
|
2895
|
+
function isPaidCallControlError$1(error) {
|
|
2860
2896
|
return error instanceof CostAccountingIncompleteError || error instanceof CostCallConflictError || error instanceof CostCeilingReachedError || error instanceof CostLedgerPersistenceError || error instanceof CostReceiptCaptureError || error instanceof CostReservationExceededError;
|
|
2861
2897
|
}
|
|
2862
2898
|
function receiptForProviderFailure(error, completedResult, model) {
|
|
@@ -2973,7 +3009,7 @@ async function publicBenchmarkPredictionsToFindings(options) {
|
|
|
2973
3009
|
evidence_refs: [evidenceByStep.get(step)],
|
|
2974
3010
|
recommended_action: prediction.recommended_action,
|
|
2975
3011
|
metadata: {
|
|
2976
|
-
analysis_mode: "
|
|
3012
|
+
analysis_mode: "direct-baseline",
|
|
2977
3013
|
model: options.providerModel
|
|
2978
3014
|
},
|
|
2979
3015
|
produced_at: options.producedAt,
|
|
@@ -3067,11 +3103,174 @@ Each finding's step MUST be the positive integer n from an existing assistant LL
|
|
|
3067
3103
|
Never select an EVALUATOR, TOOL, CHAIN, final-verification, benchmark-verification, or message-<n> span.
|
|
3068
3104
|
Before emitting a finding, inspect its candidate span's attributes.content and describe only the action shown there.
|
|
3069
3105
|
When the trajectory has no incorrect steps, return an empty findings array.`;
|
|
3106
|
+
function trajectoryIdFromCaseId$1(dataset, caseId) {
|
|
3107
|
+
const prefix = dataset === "agentrx" ? "agentrx:" : "codetrace:";
|
|
3108
|
+
if (!caseId.startsWith(prefix) || caseId.length === prefix.length) throw new Error(`unexpected ${dataset} benchmark case id '${caseId}'`);
|
|
3109
|
+
return caseId.slice(prefix.length);
|
|
3110
|
+
}
|
|
3111
|
+
//#endregion
|
|
3112
|
+
//#region src/analyst/benchmark-public-rlm.ts
|
|
3113
|
+
const AGENT_RX_RLM_INSTRUCTIONS = `Analyze exactly one failed agent trajectory.
|
|
3114
|
+
Find the first unrecoverable critical failure, not every later symptom.
|
|
3115
|
+
Use the trace tools to inspect the action and its following observation.
|
|
3116
|
+
Emit zero findings only when evidence does not support a root cause.
|
|
3117
|
+
Otherwise emit exactly one finding whose subject is exactly one of:
|
|
3118
|
+
instruction-plan-adherence-failure
|
|
3119
|
+
invention-of-new-information
|
|
3120
|
+
invalid-invocation
|
|
3121
|
+
misinterpretation-of-tool-output-handoff-failure
|
|
3122
|
+
intent-plan-misalignment
|
|
3123
|
+
underspecified-user-intent
|
|
3124
|
+
intent-not-supported
|
|
3125
|
+
guardrails-triggered
|
|
3126
|
+
system-failure
|
|
3127
|
+
inconclusive
|
|
3128
|
+
Cite exactly one assistant span named step-<n> as trace://<trace-id>/span/step-<n>.
|
|
3129
|
+
The excerpt must quote the assistant action exactly.`;
|
|
3130
|
+
const CODE_TRACE_RLM_INSTRUCTIONS = `${CODE_TRACE_BENCH_ANALYST_PROMPT}
|
|
3131
|
+
Use the trace tools rather than asking for the whole trajectory in the prompt.
|
|
3132
|
+
Keep retrieved trace objects in Python variables.
|
|
3133
|
+
Never print an entire trace, full source file, or more than 12000 characters in one iteration.
|
|
3134
|
+
Build a compact table of assistant step ids, actions, following observations, and final verification.
|
|
3135
|
+
Inspect suspicious steps with viewSpans or searchSpan instead of repeatedly printing the table.
|
|
3136
|
+
Submit as soon as every state-changing assistant step has a supported verdict.
|
|
3137
|
+
For each incorrect step, set subject to incorrect-step-<n>.
|
|
3138
|
+
Cite the assistant span as trace://<URL-encoded-trace-id>/span/step-<n>.
|
|
3139
|
+
The evidence excerpt must be an exact quote from that span's action content.
|
|
3140
|
+
Return no finding for a clean trajectory.`;
|
|
3141
|
+
/** Public benchmark candidate that runs the actual recursive trace analyst. */
|
|
3142
|
+
function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
3143
|
+
const costLedger = config.costLedger ?? new CostLedger();
|
|
3144
|
+
const limits = {
|
|
3145
|
+
maxIterations: config.dspyRlm?.maxIterations ?? 8,
|
|
3146
|
+
maxLlmCalls: config.dspyRlm?.maxLlmCalls ?? 4,
|
|
3147
|
+
maxToolCalls: config.dspyRlm?.maxToolCalls ?? 32,
|
|
3148
|
+
maxOutputChars: config.dspyRlm?.maxOutputChars ?? 8e3
|
|
3149
|
+
};
|
|
3150
|
+
const pricing = config.pricing ?? pricingForModel(config.model);
|
|
3151
|
+
const engine = createDspyRlmTraceEngine({
|
|
3152
|
+
baseUrl: config.baseUrl,
|
|
3153
|
+
apiKey: config.apiKey,
|
|
3154
|
+
model: config.model,
|
|
3155
|
+
maxOutputTokens: config.maxOutputTokens,
|
|
3156
|
+
timeoutMs: config.timeoutMs,
|
|
3157
|
+
maxCostUsd: config.maxCostUsdPerAnalysis ?? 1,
|
|
3158
|
+
pricing,
|
|
3159
|
+
...config.dspyRlm?.runner ? { runner: config.dspyRlm.runner } : {}
|
|
3160
|
+
});
|
|
3161
|
+
const definition = publicBenchmarkDefinition(dataset, limits);
|
|
3162
|
+
return {
|
|
3163
|
+
id: "dspy-rlm",
|
|
3164
|
+
async analyze(input, context) {
|
|
3165
|
+
const trajectoryId = trajectoryIdFromCaseId(dataset, context.caseId);
|
|
3166
|
+
const tags = {
|
|
3167
|
+
benchmarkCaseId: context.caseId,
|
|
3168
|
+
benchmarkRepetition: String(context.repetition)
|
|
3169
|
+
};
|
|
3170
|
+
let usage;
|
|
3171
|
+
let rawFindings = [];
|
|
3172
|
+
try {
|
|
3173
|
+
if (!input.traceStore) throw new Error(`${dataset} DSPy RLM runner requires a trace store`);
|
|
3174
|
+
const completed = await runTraceAnalyst({
|
|
3175
|
+
definition,
|
|
3176
|
+
engine,
|
|
3177
|
+
store: input.traceStore,
|
|
3178
|
+
context: {
|
|
3179
|
+
runId: context.caseId,
|
|
3180
|
+
correlationId: `${context.caseId}:${context.repetition}`,
|
|
3181
|
+
costLedger,
|
|
3182
|
+
costPhase: "analyst.public-benchmark.dspy-rlm",
|
|
3183
|
+
tags,
|
|
3184
|
+
recordUsage: (receipt) => {
|
|
3185
|
+
usage = receipt;
|
|
3186
|
+
},
|
|
3187
|
+
signal: context.signal
|
|
3188
|
+
}
|
|
3189
|
+
});
|
|
3190
|
+
const producedAt = (/* @__PURE__ */ new Date()).toISOString();
|
|
3191
|
+
rawFindings = completed.findings.map((finding) => makeFinding({
|
|
3192
|
+
analyst_id: "dspy-rlm",
|
|
3193
|
+
area: dataset === "agentrx" ? "root-cause" : "incorrect",
|
|
3194
|
+
subject: finding.subject,
|
|
3195
|
+
claim: finding.claim,
|
|
3196
|
+
rationale: finding.rationale,
|
|
3197
|
+
severity: finding.severity,
|
|
3198
|
+
confidence: finding.confidence,
|
|
3199
|
+
evidence_refs: evidenceRefsFromRawFinding(finding),
|
|
3200
|
+
recommended_action: finding.recommended_action,
|
|
3201
|
+
metadata: {
|
|
3202
|
+
analysis_mode: "recursive",
|
|
3203
|
+
engine: "dspy-rlm",
|
|
3204
|
+
model: config.model
|
|
3205
|
+
},
|
|
3206
|
+
produced_at: producedAt
|
|
3207
|
+
}));
|
|
3208
|
+
const findings = adaptPublicBenchmarkFindings(dataset, trajectoryId, rawFindings, "dspy-rlm");
|
|
3209
|
+
if (dataset === "codetracebench") await validateCodeTraceFindingEvidence({
|
|
3210
|
+
trajectoryId,
|
|
3211
|
+
findings,
|
|
3212
|
+
store: input.traceStore,
|
|
3213
|
+
...context.signal ? { signal: context.signal } : {}
|
|
3214
|
+
});
|
|
3215
|
+
return {
|
|
3216
|
+
findings,
|
|
3217
|
+
usage,
|
|
3218
|
+
metadata: {
|
|
3219
|
+
analysisMode: "recursive",
|
|
3220
|
+
engine: "dspy-rlm",
|
|
3221
|
+
protocolSha256: publicBenchmarkProtocolSha256(dataset),
|
|
3222
|
+
answer: completed.answer,
|
|
3223
|
+
trajectory: completed.trajectory,
|
|
3224
|
+
modelCalls: completed.modelCalls,
|
|
3225
|
+
toolCalls: completed.toolCalls,
|
|
3226
|
+
runtime: completed.runtime
|
|
3227
|
+
}
|
|
3228
|
+
};
|
|
3229
|
+
} catch (error) {
|
|
3230
|
+
if (context.signal?.aborted) throw error;
|
|
3231
|
+
if (isPaidCallControlError(error)) throw error;
|
|
3232
|
+
return {
|
|
3233
|
+
findings: [],
|
|
3234
|
+
usage,
|
|
3235
|
+
error: publicBenchmarkError(error, [config.apiKey]),
|
|
3236
|
+
metadata: {
|
|
3237
|
+
analysisMode: "recursive",
|
|
3238
|
+
engine: "dspy-rlm",
|
|
3239
|
+
rawFindings
|
|
3240
|
+
}
|
|
3241
|
+
};
|
|
3242
|
+
}
|
|
3243
|
+
}
|
|
3244
|
+
};
|
|
3245
|
+
}
|
|
3246
|
+
function publicBenchmarkDefinition(dataset, limits) {
|
|
3247
|
+
return {
|
|
3248
|
+
id: dataset === "agentrx" ? "agentrx-dspy-rlm" : "codetracebench-dspy-rlm",
|
|
3249
|
+
description: dataset === "agentrx" ? "Localizes the first unrecoverable root-cause step." : "Localizes every incorrect state-changing assistant step.",
|
|
3250
|
+
area: dataset === "agentrx" ? "root-cause" : "incorrect",
|
|
3251
|
+
version: "1.0.0",
|
|
3252
|
+
question: dataset === "agentrx" ? "What is the first unrecoverable root cause in this failed trajectory?" : "Which assistant steps are incorrect under the CodeTraceBench definition?",
|
|
3253
|
+
instructions: dataset === "agentrx" ? AGENT_RX_RLM_INSTRUCTIONS : CODE_TRACE_RLM_INSTRUCTIONS,
|
|
3254
|
+
toolGroup: "singleTrace",
|
|
3255
|
+
limits
|
|
3256
|
+
};
|
|
3257
|
+
}
|
|
3258
|
+
function pricingForModel(model) {
|
|
3259
|
+
const pricing = resolveModelPricing(model);
|
|
3260
|
+
if (!pricing) throw new Error(`no pricing is configured for '${model}'; provide PublicAnalystBenchmarkModelConfig.pricing`);
|
|
3261
|
+
return {
|
|
3262
|
+
inputUsdPerMillion: pricing.input * 1e3,
|
|
3263
|
+
outputUsdPerMillion: pricing.output * 1e3
|
|
3264
|
+
};
|
|
3265
|
+
}
|
|
3070
3266
|
function trajectoryIdFromCaseId(dataset, caseId) {
|
|
3071
3267
|
const prefix = dataset === "agentrx" ? "agentrx:" : "codetrace:";
|
|
3072
3268
|
if (!caseId.startsWith(prefix) || caseId.length === prefix.length) throw new Error(`unexpected ${dataset} benchmark case id '${caseId}'`);
|
|
3073
3269
|
return caseId.slice(prefix.length);
|
|
3074
3270
|
}
|
|
3271
|
+
function isPaidCallControlError(error) {
|
|
3272
|
+
return error instanceof CostAccountingIncompleteError || error instanceof CostCallConflictError || error instanceof CostCeilingReachedError || error instanceof CostLedgerPersistenceError || error instanceof CostReceiptCaptureError || error instanceof CostReservationExceededError;
|
|
3273
|
+
}
|
|
3075
3274
|
//#endregion
|
|
3076
3275
|
//#region src/analyst/benchmark-command-persistence.ts
|
|
3077
3276
|
const ANALYST_BENCHMARK_INITIALIZATION_COMPLETE_FILE = "initialization-complete.json";
|
|
@@ -3142,7 +3341,7 @@ function createRunIdentity(config, prepared) {
|
|
|
3142
3341
|
analystProtocolSha256: publicBenchmarkProtocolSha256(config.dataset),
|
|
3143
3342
|
implementationSha256: ANALYST_BENCHMARK_IMPLEMENTATION_SHA256,
|
|
3144
3343
|
dependencyLockSha256: ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256,
|
|
3145
|
-
runnerIds: ["empty", "
|
|
3344
|
+
runnerIds: ["empty", "dspy-rlm"]
|
|
3146
3345
|
},
|
|
3147
3346
|
inputs: {
|
|
3148
3347
|
labelsSha256: prepared.labelsSha256,
|
|
@@ -3347,7 +3546,7 @@ async function readProgress(path, runIdentitySha256, caseIds, repetitions) {
|
|
|
3347
3546
|
previousRowSha256: parsed.previousRowSha256,
|
|
3348
3547
|
observation
|
|
3349
3548
|
}) !== parsed.rowSha256) throw new Error(`benchmark observation row ${index + 1} digest does not match its contents`);
|
|
3350
|
-
if (!allowedCases.has(observation.caseId) || observation.runnerId !== "empty" && observation.runnerId !== "
|
|
3549
|
+
if (!allowedCases.has(observation.caseId) || observation.runnerId !== "empty" && observation.runnerId !== "dspy-rlm" || observation.repetition >= repetitions || observation.executionIndex >= plannedObservationCount) throw new Error(`benchmark observation row ${index + 1} does not match a planned case, runner, and repetition`);
|
|
3351
3550
|
if (executionIndexes.has(observation.executionIndex)) throw new Error(`duplicate benchmark executionIndex ${observation.executionIndex} at line ${index + 1}`);
|
|
3352
3551
|
observations.push(observation);
|
|
3353
3552
|
seen.add(key);
|
|
@@ -3802,7 +4001,7 @@ function assertCompletedArtifactMatchesRun(artifact, manifest, observations, pre
|
|
|
3802
4001
|
if (canonicalJson(artifact.result.summaries) !== canonicalJson(expectedSummaries)) throw new Error("completed benchmark summaries do not match durable observations");
|
|
3803
4002
|
const expectedComparisons = [compareAnalystRunners(artifact.result, {
|
|
3804
4003
|
baselineRunnerId: "empty",
|
|
3805
|
-
candidateRunnerId: "
|
|
4004
|
+
candidateRunnerId: "dspy-rlm",
|
|
3806
4005
|
seed: config.seed
|
|
3807
4006
|
})];
|
|
3808
4007
|
if (canonicalJson(artifact.comparisons) !== canonicalJson(expectedComparisons)) throw new Error("completed benchmark comparisons do not match durable observations");
|
|
@@ -3951,8 +4150,8 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
|
|
|
3951
4150
|
return benchmarkExitCode(artifact.result);
|
|
3952
4151
|
}
|
|
3953
4152
|
if (await regularFileExists(paths.report)) throw new Error(`benchmark report exists without a completed result; refusing ambiguous resume: ${paths.report}`);
|
|
3954
|
-
const
|
|
3955
|
-
const runners = [emptyPublicBenchmarkRunner(),
|
|
4153
|
+
const createAnalystRunner = dependencies.createAnalystRunner ?? ((dataset, model) => createPublicBenchmarkRlmRunner(dataset, model));
|
|
4154
|
+
const runners = [emptyPublicBenchmarkRunner(), createAnalystRunner(config.dataset, {
|
|
3956
4155
|
...config.model,
|
|
3957
4156
|
costLedger,
|
|
3958
4157
|
durability: {
|
|
@@ -4016,7 +4215,7 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
|
|
|
4016
4215
|
assertSameObservations(result.observations, persisted.observations);
|
|
4017
4216
|
const comparisons = [compareAnalystRunners(result, {
|
|
4018
4217
|
baselineRunnerId: "empty",
|
|
4019
|
-
candidateRunnerId: "
|
|
4218
|
+
candidateRunnerId: "dspy-rlm",
|
|
4020
4219
|
seed: config.seed
|
|
4021
4220
|
})];
|
|
4022
4221
|
const codeTraceCalibration = config.dataset === "codetracebench" ? summarizeCodeTraceCalibration(result) : void 0;
|
|
@@ -4073,7 +4272,7 @@ const NON_SCORABLE_COST_ERRORS = /* @__PURE__ */ new Set([
|
|
|
4073
4272
|
]);
|
|
4074
4273
|
function assertObservationAccountingComplete(observation, costLedger) {
|
|
4075
4274
|
if (observation.error && NON_SCORABLE_COST_ERRORS.has(observation.error.class)) throw new CostAccountingIncompleteError(`Analyst benchmark stopped before scoring: ${observation.error.message}`);
|
|
4076
|
-
if (observation.runnerId !== "
|
|
4275
|
+
if (observation.runnerId !== "dspy-rlm") return;
|
|
4077
4276
|
const summary = costLedger.summary({
|
|
4078
4277
|
channel: "analyst",
|
|
4079
4278
|
tags: {
|
|
@@ -4081,7 +4280,7 @@ function assertObservationAccountingComplete(observation, costLedger) {
|
|
|
4081
4280
|
benchmarkRepetition: String(observation.repetition)
|
|
4082
4281
|
}
|
|
4083
4282
|
});
|
|
4084
|
-
if (!summary.accountingComplete || summary.pendingCalls > 0 || summary.unresolvedCalls > 0) throw accountingError(costLedger, "the
|
|
4283
|
+
if (!summary.accountingComplete || summary.pendingCalls > 0 || summary.unresolvedCalls > 0) throw accountingError(costLedger, "the recursive analyst has incomplete cost accounting", {
|
|
4085
4284
|
channel: "analyst",
|
|
4086
4285
|
tags: {
|
|
4087
4286
|
benchmarkCaseId: observation.caseId,
|
|
@@ -4099,7 +4298,7 @@ function accountingError(costLedger, reason, filter) {
|
|
|
4099
4298
|
}
|
|
4100
4299
|
const ANALYST_BENCHMARK_HELP = `agent-eval analyst-benchmark
|
|
4101
4300
|
|
|
4102
|
-
Run
|
|
4301
|
+
Run the recursive DSPy trace analyst against public AgentRx or CodeTraceBench labels.
|
|
4103
4302
|
|
|
4104
4303
|
Required:
|
|
4105
4304
|
--dataset agentrx|codetracebench
|
|
@@ -4120,6 +4319,7 @@ Controls:
|
|
|
4120
4319
|
--concurrency <positive integer> Parallel benchmark jobs. Default: 1
|
|
4121
4320
|
--repetitions <positive integer> Runs per case and runner. Default: 1
|
|
4122
4321
|
--max-output-tokens <positive> Model output limit per call. Default: 4096
|
|
4322
|
+
--python <executable> Python with agent-eval-rpc[dspy]. Default: python
|
|
4123
4323
|
--timeout-ms <positive> Model analyst deadline per case. Default: 300000
|
|
4124
4324
|
--max-cost-usd <positive> Run-wide spend limit. Default: 5
|
|
4125
4325
|
--max-artifact-bytes <positive> Final evidence bytes per case. Default: ${DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES}
|
|
@@ -4142,6 +4342,8 @@ function parseCommandConfig(argv, env) {
|
|
|
4142
4342
|
if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(apiKeyEnv)) throw new Error("--api-key-env must be a valid environment variable name");
|
|
4143
4343
|
const apiKey = env[apiKeyEnv]?.trim();
|
|
4144
4344
|
if (!apiKey) throw new Error(`--api-key-env points to an empty or missing variable: ${apiKeyEnv}`);
|
|
4345
|
+
const maxCostUsd = positiveFiniteFlag(flags, "max-cost-usd", 5);
|
|
4346
|
+
const python = flags.get("python")?.trim();
|
|
4145
4347
|
return {
|
|
4146
4348
|
dataset,
|
|
4147
4349
|
labelsPath: requiredFlag(flags, "labels"),
|
|
@@ -4155,13 +4357,15 @@ function parseCommandConfig(argv, env) {
|
|
|
4155
4357
|
apiKey,
|
|
4156
4358
|
model: requiredFlag(flags, "model"),
|
|
4157
4359
|
maxOutputTokens: positiveFlag(flags, "max-output-tokens", 4096),
|
|
4158
|
-
timeoutMs: positiveFlag(flags, "timeout-ms", 3e5)
|
|
4360
|
+
timeoutMs: positiveFlag(flags, "timeout-ms", 3e5),
|
|
4361
|
+
maxCostUsdPerAnalysis: maxCostUsd,
|
|
4362
|
+
...python ? { dspyRlm: { runner: { command: python } } } : {}
|
|
4159
4363
|
},
|
|
4160
4364
|
limit: positiveFlag(flags, "limit"),
|
|
4161
4365
|
seed: integerFlag(flags, "seed", 0),
|
|
4162
4366
|
concurrency: positiveFlag(flags, "concurrency", 1),
|
|
4163
4367
|
repetitions: positiveFlag(flags, "repetitions", 1),
|
|
4164
|
-
maxCostUsd
|
|
4368
|
+
maxCostUsd,
|
|
4165
4369
|
maxArtifactBytes: positiveFlag(flags, "max-artifact-bytes", DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES),
|
|
4166
4370
|
apiKeyEnv,
|
|
4167
4371
|
command: `agent-eval analyst-benchmark ${argv.filter((argument) => argument !== "--resume").map(shellQuote).join(" ")}`,
|
|
@@ -4206,6 +4410,7 @@ const KNOWN_FLAGS = /* @__PURE__ */ new Set([
|
|
|
4206
4410
|
"concurrency",
|
|
4207
4411
|
"repetitions",
|
|
4208
4412
|
"max-output-tokens",
|
|
4413
|
+
"python",
|
|
4209
4414
|
"timeout-ms",
|
|
4210
4415
|
"max-cost-usd",
|
|
4211
4416
|
"max-artifact-bytes"
|
|
@@ -4315,7 +4520,7 @@ function renderArtifactMarkdown(artifact) {
|
|
|
4315
4520
|
return `${renderAnalystBenchmarkMarkdown(artifact.result, artifact.comparisons).trimEnd()}${calibrationMarkdown}${verificationMarkdown}\n\n${renderSelectionMarkdown(artifact.inputs.selection.report)}\n`;
|
|
4316
4521
|
}
|
|
4317
4522
|
function benchmarkExitCode(result) {
|
|
4318
|
-
return result.summaries.find((summary) => summary.runnerId === "
|
|
4523
|
+
return result.summaries.find((summary) => summary.runnerId === "dspy-rlm")?.failedRuns ? 2 : 0;
|
|
4319
4524
|
}
|
|
4320
4525
|
function printSuccessSummary(artifact, paths) {
|
|
4321
4526
|
const failures = artifact.result.summaries.reduce((total, summary) => total + summary.failedRuns, 0);
|
|
@@ -4327,6 +4532,6 @@ function shellQuote(value) {
|
|
|
4327
4532
|
return /^[a-zA-Z0-9_./:@%+=,-]+$/.test(value) ? value : `'${value.replaceAll("'", "'\\''")}'`;
|
|
4328
4533
|
}
|
|
4329
4534
|
//#endregion
|
|
4330
|
-
export {
|
|
4535
|
+
export { ANALYST_BENCHMARK_IMPLEMENTATION_FILES as A, summarizeAgentRxCalibration as B, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as C, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as D, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as E, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as F, normalizeAgentRxCategory as G, codeTracerPredictionsToFindings as H, ANALYST_BENCHMARK_MANIFEST_FILE as I, roundAgentRxStep as K, ANALYST_BENCHMARK_OBSERVATIONS_FILE as L, analystBenchmarkDependencyLockDigest as M, analystBenchmarkImplementationDigest as N, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as O, ANALYST_BENCHMARK_COST_LEDGER_FILE as P, AGENT_RX_UPSTREAM_REVISION as R, emptyPublicBenchmarkRunner as S, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as T, agentRxBenchmarkCase as U, codeTraceBenchCase as V, agentRxPredictionsToFindings as W, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as _, renderCodeTraceCalibrationMarkdown as a, parseVerificationOutcome as b, createPublicBenchmarkRlmRunner as c, publicBenchmarkProtocolSha256 as d, loadPublicBenchmarkRows as f, selectPublicBenchmarkRows as g, publicBenchmarkSelectionReport as h, readAnalystBenchmarkArtifact as i, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as j, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as k, CODE_TRACE_BENCH_ANALYST_PROMPT as l, publicBenchmarkDistributions as m, runAnalystBenchmarkCommand as n, summarizeCodeTraceCalibration as o, preparePublicAnalystBenchmark as p, normalizeBenchmarkLabel as q, renderAnalystBenchmarkMarkdown as r, compareAnalystRunners as s, ANALYST_BENCHMARK_HELP as t, createPublicBenchmarkDirectRunner as u, appendVerificationArtifactsToOtlp as v, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as w, adaptPublicBenchmarkFindings as x, loadCodeTraceVerificationArtifacts as y, renderAgentRxCalibrationMarkdown as z };
|
|
4331
4536
|
|
|
4332
|
-
//# sourceMappingURL=benchmark-command-
|
|
4537
|
+
//# sourceMappingURL=benchmark-command-BKfjOBJ5.js.map
|