@tangle-network/agent-eval 0.139.2 → 0.140.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +22 -0
- package/dist/analyst/index.d.ts +112 -18
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +7 -7
- package/dist/{benchmark-DDVdWcwA.d.ts → benchmark-DxaZfy0w.d.ts} +3 -3
- package/dist/{benchmark-DDVdWcwA.d.ts.map → benchmark-DxaZfy0w.d.ts.map} +1 -1
- package/dist/{benchmark-command-DzUZJl8M.js → benchmark-command-CK0UnXAD.js} +633 -339
- package/dist/benchmark-command-CK0UnXAD.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-zxhy1QV3.js → benchmarks-HwoBE32G.js} +4 -4
- package/dist/{benchmarks-zxhy1QV3.js.map → benchmarks-HwoBE32G.js.map} +1 -1
- package/dist/campaign/index.d.ts +5 -5
- package/dist/campaign/index.js +3 -3
- package/dist/{campaign-DrS6_hLd.js → campaign-BzjYNYVZ.js} +5 -5
- package/dist/{campaign-DrS6_hLd.js.map → campaign-BzjYNYVZ.js.map} +1 -1
- package/dist/cli.js +2 -2
- package/dist/{client-BohnDFBq.d.ts → client-BoqGxEqx.d.ts} +4 -4
- package/dist/{client-BohnDFBq.d.ts.map → client-BoqGxEqx.d.ts.map} +1 -1
- package/dist/{completion-verifier-IPoP4fQO.d.ts → completion-verifier-D15NHYSk.d.ts} +5 -5
- package/dist/{completion-verifier-IPoP4fQO.d.ts.map → completion-verifier-D15NHYSk.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +10 -10
- package/dist/contract/index.js +6 -6
- package/dist/control.d.ts +2 -2
- package/dist/{cost-ledger-CZ9diLxY.js → cost-ledger-DMFxsLKr.js} +22 -8
- package/dist/cost-ledger-DMFxsLKr.js.map +1 -0
- package/dist/{cost-ledger-DKgyIWRj.d.ts → cost-ledger-FuQvHxPm.d.ts} +8 -2
- package/dist/{cost-ledger-DKgyIWRj.d.ts.map → cost-ledger-FuQvHxPm.d.ts.map} +1 -1
- package/dist/{default-registry-B8vf7Rmf.d.ts → default-registry-Ci7wAAR8.d.ts} +5 -5
- package/dist/{default-registry-B8vf7Rmf.d.ts.map → default-registry-Ci7wAAR8.d.ts.map} +1 -1
- package/dist/{default-registry-BgJJItGr.js → default-registry-DCp-6hc-.js} +3 -3
- package/dist/{default-registry-BgJJItGr.js.map → default-registry-DCp-6hc-.js.map} +1 -1
- package/dist/{dspy-rlm-engine-DTkVyDX-.js → dspy-rlm-engine-Bkak4nzo.js} +11 -4
- package/dist/dspy-rlm-engine-Bkak4nzo.js.map +1 -0
- package/dist/{eval-campaign-BmptJj50.js → eval-campaign-YdkpWWoT.js} +2 -2
- package/dist/{eval-campaign-BmptJj50.js.map → eval-campaign-YdkpWWoT.js.map} +1 -1
- package/dist/{exact-types-MaaFcllV.d.ts → exact-types-B0lJV3tu.d.ts} +2 -2
- package/dist/{exact-types-MaaFcllV.d.ts.map → exact-types-B0lJV3tu.d.ts.map} +1 -1
- package/dist/{external-optimizer-contracts-BrxY2Sli.d.ts → external-optimizer-contracts-nb7c_WAR.d.ts} +12 -2
- package/dist/{external-optimizer-contracts-BrxY2Sli.d.ts.map → external-optimizer-contracts-nb7c_WAR.d.ts.map} +1 -1
- package/dist/{extract-usage-DZs601Va.js → extract-usage-C5vMw-0R.js} +2 -2
- package/dist/{extract-usage-DZs601Va.js.map → extract-usage-C5vMw-0R.js.map} +1 -1
- package/dist/{feedback-trajectory-BJUWOkJM.d.ts → feedback-trajectory-BCHqzLh3.d.ts} +3 -3
- package/dist/{feedback-trajectory-BJUWOkJM.d.ts.map → feedback-trajectory-BCHqzLh3.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BipJlj-C.d.ts → index-B798zyGh.d.ts} +5 -2
- package/dist/{index-BipJlj-C.d.ts.map → index-B798zyGh.d.ts.map} +1 -1
- package/dist/{index-_66rVpwN.d.ts → index-CKI1CXTL.d.ts} +5 -5
- package/dist/{index-_66rVpwN.d.ts.map → index-CKI1CXTL.d.ts.map} +1 -1
- package/dist/{index-CWOPCJiw.d.ts → index-D2enbqA0.d.ts} +2 -2
- package/dist/{index-CWOPCJiw.d.ts.map → index-D2enbqA0.d.ts.map} +1 -1
- package/dist/{index-BTm_P9aC.d.ts → index-DCP4I2Qx.d.ts} +10 -10
- package/dist/{index-BTm_P9aC.d.ts.map → index-DCP4I2Qx.d.ts.map} +1 -1
- package/dist/{index-CtR1xh4V.d.ts → index-DFLVtPZ9.d.ts} +3 -3
- package/dist/{index-CtR1xh4V.d.ts.map → index-DFLVtPZ9.d.ts.map} +1 -1
- package/dist/index.d.ts +24 -24
- package/dist/index.js +14 -14
- package/dist/{insight-report-Bu5Wi9tG.d.ts → insight-report-Bh_8ksel.d.ts} +4 -4
- package/dist/{insight-report-Bu5Wi9tG.d.ts.map → insight-report-Bh_8ksel.d.ts.map} +1 -1
- package/dist/{integrity-COTh3DTH.d.ts → integrity-DRXobPEs.d.ts} +2 -2
- package/dist/{integrity-COTh3DTH.d.ts.map → integrity-DRXobPEs.d.ts.map} +1 -1
- package/dist/{kind-factory-CFxA0JQX.js → kind-factory-DB7nIs35.js} +2 -2
- package/dist/{kind-factory-CFxA0JQX.js.map → kind-factory-DB7nIs35.js.map} +1 -1
- package/dist/{llm-client-bkztEfIx.js → llm-client-B3WXSH5Y.js} +2 -2
- package/dist/{llm-client-bkztEfIx.js.map → llm-client-B3WXSH5Y.js.map} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{release-report-fZarvIm-.d.ts → release-report-B_bQOHM-.d.ts} +3 -3
- package/dist/{release-report-fZarvIm-.d.ts.map → release-report-B_bQOHM-.d.ts.map} +1 -1
- package/dist/{replay-DjG4IG60.d.ts → replay-BqTgoioO.d.ts} +6 -6
- package/dist/{replay-DjG4IG60.d.ts.map → replay-BqTgoioO.d.ts.map} +1 -1
- package/dist/{replay-SA4OB7O7.js → replay-k2MsOmv5.js} +4 -4
- package/dist/{replay-SA4OB7O7.js.map → replay-k2MsOmv5.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-BxhtGfKa.d.ts → researcher-C6lzl-rP.d.ts} +5 -5
- package/dist/{researcher-BxhtGfKa.d.ts.map → researcher-C6lzl-rP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CqSLiV51.d.ts → reward-hacking-CEVVmy3h.d.ts} +2 -2
- package/dist/{reward-hacking-CqSLiV51.d.ts.map → reward-hacking-CEVVmy3h.d.ts.map} +1 -1
- package/dist/rl.d.ts +5 -5
- package/dist/rl.js +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/{rubric-predictive-validity-DQBQj6uV.d.ts → rubric-predictive-validity-C2CthIfY.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-DQBQj6uV.d.ts.map → rubric-predictive-validity-C2CthIfY.d.ts.map} +1 -1
- package/dist/{run-evidence-C4RcRQT5.d.ts → run-evidence-8Ou28QSa.d.ts} +3 -3
- package/dist/{run-evidence-C4RcRQT5.d.ts.map → run-evidence-8Ou28QSa.d.ts.map} +1 -1
- package/dist/{run-record-CztDMXVF.d.ts → run-record-Tb3TTtUn.d.ts} +2 -2
- package/dist/{run-record-CztDMXVF.d.ts.map → run-record-Tb3TTtUn.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-BuIJ9IfB.js → semantic-concept-judge-DJQtFr95.js} +3 -3
- package/dist/{semantic-concept-judge-BuIJ9IfB.js.map → semantic-concept-judge-DJQtFr95.js.map} +1 -1
- package/dist/{server-DaCpLfi0.js → server-Cu4M3NSO.js} +3 -3
- package/dist/{server-DaCpLfi0.js.map → server-Cu4M3NSO.js.map} +1 -1
- package/dist/{single-run-lock-BTTtPZ9N.js → single-run-lock-CiQThJxB.js} +22 -12
- package/dist/single-run-lock-CiQThJxB.js.map +1 -0
- package/dist/{skill-usage-B-BFS8M2.d.ts → skill-usage-CVVnoIx-.d.ts} +26 -10
- package/dist/skill-usage-CVVnoIx-.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-BbGnCC53.js → skillopt-optimization-method-CSBQ8Qma.js} +4 -4
- package/dist/{skillopt-optimization-method-BbGnCC53.js.map → skillopt-optimization-method-CSBQ8Qma.js.map} +1 -1
- package/dist/{skillopt-optimization-method-_s0Tub7Y.d.ts → skillopt-optimization-method-D1dqGzzH.d.ts} +11 -11
- package/dist/{skillopt-optimization-method-_s0Tub7Y.d.ts.map → skillopt-optimization-method-D1dqGzzH.d.ts.map} +1 -1
- package/dist/{statistics-B5d0Zd-z.d.ts → statistics-B4u_CiFd.d.ts} +2 -2
- package/dist/{statistics-B5d0Zd-z.d.ts.map → statistics-B4u_CiFd.d.ts.map} +1 -1
- package/dist/{store-otlp-DX4fGIcf.js → store-otlp-vRByAR6h.js} +2 -2
- package/dist/{store-otlp-DX4fGIcf.js.map → store-otlp-vRByAR6h.js.map} +1 -1
- package/dist/{summary-report-Cg7BifAM.d.ts → summary-report-o3eJ3gxG.d.ts} +3 -3
- package/dist/{summary-report-Cg7BifAM.d.ts.map → summary-report-o3eJ3gxG.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +1 -1
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-B2EWUmQY.js → supervisor-run-D6A5oQw-.js} +18 -9
- package/dist/supervisor-run-D6A5oQw-.js.map +1 -0
- package/dist/{tool-groups-CdYq22lX.d.ts → tool-groups-DVQTy9lq.d.ts} +8 -8
- package/dist/{tool-groups-CdYq22lX.d.ts.map → tool-groups-DVQTy9lq.d.ts.map} +1 -1
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +4 -4
- package/dist/{types-uPrS6mD-.d.ts → types-BjMFz88h.d.ts} +2 -2
- package/dist/{types-uPrS6mD-.d.ts.map → types-BjMFz88h.d.ts.map} +1 -1
- package/dist/{types-DoEYskCd.d.ts → types-D3jh6F98.d.ts} +4 -4
- package/dist/{types-DoEYskCd.d.ts.map → types-D3jh6F98.d.ts.map} +1 -1
- package/dist/{types-BBFNHxSK.d.ts → types-Dk7PB7vh.d.ts} +5 -5
- package/dist/{types-BBFNHxSK.d.ts.map → types-Dk7PB7vh.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/package.json +1 -1
- package/dist/benchmark-command-DzUZJl8M.js.map +0 -1
- package/dist/cost-ledger-CZ9diLxY.js.map +0 -1
- package/dist/dspy-rlm-engine-DTkVyDX-.js.map +0 -1
- package/dist/single-run-lock-BTTtPZ9N.js.map +0 -1
- package/dist/skill-usage-B-BFS8M2.d.ts.map +0 -1
- package/dist/supervisor-run-B2EWUmQY.js.map +0 -1
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
import { c as ValidationError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
|
|
2
2
|
import { s as resolveModelPricing } from "./metrics-C9YY1OcL.js";
|
|
3
|
-
import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-
|
|
4
|
-
import { c as callLlmJson, f as maximumChargeForLlmRequest, l as costReceiptFromLlm, r as LlmResponseError, t as LlmCallError, u as costReceiptFromLlmError } from "./llm-client-
|
|
5
|
-
import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-
|
|
3
|
+
import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-DMFxsLKr.js";
|
|
4
|
+
import { c as callLlmJson, f as maximumChargeForLlmRequest, l as costReceiptFromLlm, r as LlmResponseError, t as LlmCallError, u as costReceiptFromLlmError } from "./llm-client-B3WXSH5Y.js";
|
|
5
|
+
import { O as evidenceRefsFromRawFinding, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-DB7nIs35.js";
|
|
6
6
|
import { o as makeFinding, r as usageReceiptFromCostLedger } from "./usage-receipt-CgxMEBZq.js";
|
|
7
7
|
import { i as hashCanonical, r as canonicalString } from "./canonical-D011XM8r.js";
|
|
8
|
-
import { n as createRunCostLedger, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-
|
|
9
|
-
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-
|
|
8
|
+
import { n as createRunCostLedger, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-CiQThJxB.js";
|
|
9
|
+
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-Bkak4nzo.js";
|
|
10
10
|
import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-Dxz0Rkwa.js";
|
|
11
11
|
import { E as pairedBootstrap } from "./statistics-ByxzSiOM.js";
|
|
12
|
-
import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-
|
|
12
|
+
import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-vRByAR6h.js";
|
|
13
13
|
import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-CYtcIF2V.js";
|
|
14
14
|
import { z } from "zod";
|
|
15
15
|
import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
|
|
@@ -1172,7 +1172,7 @@ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
|
|
|
1172
1172
|
"package.json",
|
|
1173
1173
|
"pnpm-lock.yaml"
|
|
1174
1174
|
]);
|
|
1175
|
-
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
1175
|
+
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "1e4c763f2b791c57b37c0e326c56546b433a7a99c0bc7d064abdb58b9cbf0d81";
|
|
1176
1176
|
/** The published benchmark evidence was produced at this package version, by
|
|
1177
1177
|
* the retired one-shot direct runner, before trace analysts moved to the
|
|
1178
1178
|
* recursive DSPy RLM engine. Both evidence digests below are historical facts
|
|
@@ -1203,6 +1203,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1203
1203
|
"src/analyst/benchmark-public-data.ts",
|
|
1204
1204
|
"src/analyst/benchmark-public-errors.ts",
|
|
1205
1205
|
"src/analyst/benchmark-public-model.ts",
|
|
1206
|
+
"src/analyst/benchmark-public-prompt.ts",
|
|
1206
1207
|
"src/analyst/benchmark-public-rlm.ts",
|
|
1207
1208
|
"src/analyst/benchmark-public-types.ts",
|
|
1208
1209
|
"src/analyst/benchmark-real-model.ts",
|
|
@@ -1265,7 +1266,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1265
1266
|
"src/trace/otlp-attributes.ts",
|
|
1266
1267
|
"src/trace/raw-provider-sink.ts"
|
|
1267
1268
|
]);
|
|
1268
|
-
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
1269
|
+
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "56dd4c7ed19fc5855f99ad238464ad95cb1bea155ad58c49dab3eb0f6cbe7d6a";
|
|
1269
1270
|
function analystBenchmarkImplementationDigest() {
|
|
1270
1271
|
return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256;
|
|
1271
1272
|
}
|
|
@@ -1332,19 +1333,12 @@ async function validateCodeTraceFindingEvidence(options) {
|
|
|
1332
1333
|
if (citations.length === 0) return;
|
|
1333
1334
|
for (const citation of citations) if (!citation.location || citation.location.traceId !== options.trajectoryId) throw new Error(`model finding '${citation.findingId}' cites non-case evidence '${citation.evidence.uri}'`);
|
|
1334
1335
|
const spanIds = [...new Set(citations.map((citation) => `step-${citation.location.step}`))];
|
|
1335
|
-
const spans
|
|
1336
|
-
|
|
1337
|
-
|
|
1338
|
-
|
|
1339
|
-
|
|
1340
|
-
|
|
1341
|
-
}, options.signal ? { signal: options.signal } : void 0);
|
|
1342
|
-
if (result.missing_span_ids.length > 0 || result.omitted_span_ids.length > 0) {
|
|
1343
|
-
const unavailable = [...result.missing_span_ids, ...result.omitted_span_ids];
|
|
1344
|
-
throw new Error(`model finding evidence is unavailable in the case trace: ${unavailable.join(", ")}`);
|
|
1345
|
-
}
|
|
1346
|
-
for (const span of result.spans) spans.set(span.span_id, span);
|
|
1347
|
-
}
|
|
1336
|
+
const { spans, missing } = await fetchTraceSpans(options.store, {
|
|
1337
|
+
trajectoryId: options.trajectoryId,
|
|
1338
|
+
spanIds,
|
|
1339
|
+
...options.signal ? { signal: options.signal } : {}
|
|
1340
|
+
});
|
|
1341
|
+
if (missing.length > 0) throw new Error(`model finding evidence is unavailable in the case trace: ${missing.join(", ")}`);
|
|
1348
1342
|
for (const citation of citations) {
|
|
1349
1343
|
const spanId = `step-${citation.location.step}`;
|
|
1350
1344
|
const span = spans.get(spanId);
|
|
@@ -1353,31 +1347,45 @@ async function validateCodeTraceFindingEvidence(options) {
|
|
|
1353
1347
|
assertExactActionExcerpt(citation.findingId, citation.evidence, spanId, span.attributes.content);
|
|
1354
1348
|
}
|
|
1355
1349
|
}
|
|
1350
|
+
/**
|
|
1351
|
+
* Resolve assistant-step evidence for a trajectory.
|
|
1352
|
+
*
|
|
1353
|
+
* `steps` are claims the model made explicitly: an unresolvable one is a model
|
|
1354
|
+
* error and throws. `optionalSteps` are derived by the runner (a block's
|
|
1355
|
+
* interior, a block's consequence step), so an unresolvable one is simply
|
|
1356
|
+
* absent from the returned map and the caller decides what that means.
|
|
1357
|
+
*/
|
|
1356
1358
|
async function resolveAssistantStepEvidence(options) {
|
|
1357
|
-
const
|
|
1358
|
-
|
|
1359
|
-
const
|
|
1360
|
-
const
|
|
1361
|
-
|
|
1362
|
-
|
|
1363
|
-
|
|
1364
|
-
|
|
1365
|
-
|
|
1366
|
-
|
|
1367
|
-
|
|
1368
|
-
|
|
1369
|
-
throw new Error(`model selected unavailable assistant steps: ${unavailable.join(", ")}`);
|
|
1370
|
-
}
|
|
1371
|
-
for (const span of result.spans) spans.set(span.span_id, span);
|
|
1372
|
-
}
|
|
1359
|
+
const required = [...new Set(options.steps)];
|
|
1360
|
+
const optional = [...new Set(options.optionalSteps ?? [])].filter((step) => !required.includes(step));
|
|
1361
|
+
for (const step of [...required, ...optional]) if (!Number.isSafeInteger(step) || step < 1) throw new TypeError(`assistant evidence step must be a positive safe integer: ${step}`);
|
|
1362
|
+
const steps = [...required, ...optional];
|
|
1363
|
+
if (steps.length === 0) return /* @__PURE__ */ new Map();
|
|
1364
|
+
const { spans, missing } = await fetchTraceSpans(options.store, {
|
|
1365
|
+
trajectoryId: options.trajectoryId,
|
|
1366
|
+
spanIds: steps.map((step) => `step-${step}`),
|
|
1367
|
+
...options.signal ? { signal: options.signal } : {}
|
|
1368
|
+
});
|
|
1369
|
+
const missingRequired = missing.filter((spanId) => required.some((step) => `step-${step}` === spanId));
|
|
1370
|
+
if (missingRequired.length > 0) throw new Error(`model selected unavailable assistant steps: ${missingRequired.join(", ")}`);
|
|
1373
1371
|
const evidence = /* @__PURE__ */ new Map();
|
|
1374
1372
|
for (const step of steps) {
|
|
1375
1373
|
const spanId = `step-${step}`;
|
|
1374
|
+
const optionalStep = optional.includes(step);
|
|
1376
1375
|
const span = spans.get(spanId);
|
|
1377
|
-
if (!span)
|
|
1378
|
-
|
|
1376
|
+
if (!span) {
|
|
1377
|
+
if (optionalStep) continue;
|
|
1378
|
+
throw new Error(`model selected missing assistant step '${spanId}'`);
|
|
1379
|
+
}
|
|
1380
|
+
if (span.kind !== "LLM") {
|
|
1381
|
+
if (optionalStep) continue;
|
|
1382
|
+
throw new Error(`model selected '${spanId}', which is ${span.kind}, not an assistant LLM span`);
|
|
1383
|
+
}
|
|
1379
1384
|
const content = span.attributes.content;
|
|
1380
|
-
if (typeof content !== "string" || content.trim().length === 0)
|
|
1385
|
+
if (typeof content !== "string" || content.trim().length === 0) {
|
|
1386
|
+
if (optionalStep) continue;
|
|
1387
|
+
throw new Error(`model selected '${spanId}' without action content`);
|
|
1388
|
+
}
|
|
1381
1389
|
evidence.set(step, {
|
|
1382
1390
|
kind: "span",
|
|
1383
1391
|
uri: codeTraceStepEvidenceUri(options.trajectoryId, step),
|
|
@@ -1386,6 +1394,38 @@ async function resolveAssistantStepEvidence(options) {
|
|
|
1386
1394
|
}
|
|
1387
1395
|
return evidence;
|
|
1388
1396
|
}
|
|
1397
|
+
/**
|
|
1398
|
+
* Read spans by id, paging over the store's byte-budget omissions.
|
|
1399
|
+
*
|
|
1400
|
+
* `omitted_span_ids` names spans that exist but did not fit the response
|
|
1401
|
+
* ceiling; the store guarantees at least one span lands per call, so
|
|
1402
|
+
* re-requesting exactly the omitted ids terminates. Only `missing_span_ids`
|
|
1403
|
+
* describes a span the trace does not contain.
|
|
1404
|
+
*/
|
|
1405
|
+
async function fetchTraceSpans(store, options) {
|
|
1406
|
+
const spans = /* @__PURE__ */ new Map();
|
|
1407
|
+
const missing = [];
|
|
1408
|
+
const unique = [...new Set(options.spanIds)];
|
|
1409
|
+
const context = options.signal ? { signal: options.signal } : void 0;
|
|
1410
|
+
for (let offset = 0; offset < unique.length; offset += TRACE_ANALYSIS_LIMITS.viewSpans) {
|
|
1411
|
+
let pending = unique.slice(offset, offset + TRACE_ANALYSIS_LIMITS.viewSpans);
|
|
1412
|
+
while (pending.length > 0) {
|
|
1413
|
+
const result = await store.viewSpans({
|
|
1414
|
+
trace_id: options.trajectoryId,
|
|
1415
|
+
span_ids: pending
|
|
1416
|
+
}, context);
|
|
1417
|
+
for (const span of result.spans) spans.set(span.span_id, span);
|
|
1418
|
+
missing.push(...result.missing_span_ids);
|
|
1419
|
+
const omitted = result.omitted_span_ids.filter((spanId) => !spans.has(spanId));
|
|
1420
|
+
if (omitted.length >= pending.length) throw new Error(`trace '${options.trajectoryId}' cannot project spans within the store response budget: ${omitted.join(", ")}`);
|
|
1421
|
+
pending = omitted;
|
|
1422
|
+
}
|
|
1423
|
+
}
|
|
1424
|
+
return {
|
|
1425
|
+
spans,
|
|
1426
|
+
missing
|
|
1427
|
+
};
|
|
1428
|
+
}
|
|
1389
1429
|
function scanValue(value, traceId, path, depth, serializedDepth) {
|
|
1390
1430
|
if (depth > MAX_LABEL_SCAN_DEPTH) throw new Error(`trace '${traceId}' exceeds benchmark label scan depth at ${path}`);
|
|
1391
1431
|
if (Array.isArray(value)) return 1 + value.reduce((count, entry, index) => count + scanValue(entry, traceId, `${path}[${index}]`, depth + 1, serializedDepth), 0);
|
|
@@ -1445,119 +1485,6 @@ function assertExactActionExcerpt(findingId, evidence, spanId, content) {
|
|
|
1445
1485
|
if (!content.includes(excerpt)) throw new Error(`model finding '${findingId}' excerpt is not present in '${spanId}' action content`);
|
|
1446
1486
|
}
|
|
1447
1487
|
//#endregion
|
|
1448
|
-
//#region src/analyst/benchmark-public-adapters.ts
|
|
1449
|
-
function emptyPublicBenchmarkRunner() {
|
|
1450
|
-
return {
|
|
1451
|
-
id: "empty",
|
|
1452
|
-
analyze() {
|
|
1453
|
-
return {
|
|
1454
|
-
findings: [],
|
|
1455
|
-
usage: {
|
|
1456
|
-
calls: 0,
|
|
1457
|
-
tokens: {
|
|
1458
|
-
input: 0,
|
|
1459
|
-
output: 0
|
|
1460
|
-
},
|
|
1461
|
-
cost: {
|
|
1462
|
-
kind: "observed",
|
|
1463
|
-
usd: 0
|
|
1464
|
-
}
|
|
1465
|
-
},
|
|
1466
|
-
metadata: { baseline: "emit-no-findings" }
|
|
1467
|
-
};
|
|
1468
|
-
}
|
|
1469
|
-
};
|
|
1470
|
-
}
|
|
1471
|
-
function adaptPublicBenchmarkFindings(dataset, trajectoryId, findings, analystId) {
|
|
1472
|
-
return dataset === "agentrx" ? adaptAgentRxFindings(trajectoryId, findings, analystId) : adaptCodeTraceFindings(trajectoryId, findings, analystId);
|
|
1473
|
-
}
|
|
1474
|
-
function adaptAgentRxFindings(trajectoryId, findings, analystId) {
|
|
1475
|
-
if (findings.length === 0) return [];
|
|
1476
|
-
if (findings.length !== 1) throw new Error(`AgentRx model analyst must emit zero or one root cause, received ${findings.length}`);
|
|
1477
|
-
const source = findings[0];
|
|
1478
|
-
if (!source.subject) throw new Error("AgentRx model analyst finding is missing its failure-category subject");
|
|
1479
|
-
const steps = exactFindingSteps(trajectoryId, source);
|
|
1480
|
-
if (steps.length !== 1) throw new Error(`AgentRx model analyst must cite exactly one root-cause step, received ${steps.length}`);
|
|
1481
|
-
const [adapted] = agentRxPredictionsToFindings(trajectoryId, [{
|
|
1482
|
-
failure_case: source.subject,
|
|
1483
|
-
step_number: steps[0],
|
|
1484
|
-
description: source.rationale ?? source.claim
|
|
1485
|
-
}], {
|
|
1486
|
-
analystId,
|
|
1487
|
-
producedAt: source.produced_at,
|
|
1488
|
-
confidence: source.confidence
|
|
1489
|
-
});
|
|
1490
|
-
if (!adapted) throw new Error("AgentRx output adapter produced no root-cause finding");
|
|
1491
|
-
return [{
|
|
1492
|
-
...adapted,
|
|
1493
|
-
metadata: {
|
|
1494
|
-
...adapted.metadata,
|
|
1495
|
-
sourceFindingId: source.finding_id
|
|
1496
|
-
}
|
|
1497
|
-
}];
|
|
1498
|
-
}
|
|
1499
|
-
function adaptCodeTraceFindings(trajectoryId, findings, analystId) {
|
|
1500
|
-
const clean = findings.filter((finding) => finding.subject === "clean");
|
|
1501
|
-
if (clean.length > 0) {
|
|
1502
|
-
if (findings.length !== 1) throw new Error("CodeTraceBench model analyst mixed a clean verdict with incorrect steps");
|
|
1503
|
-
exactFindingSteps(trajectoryId, clean[0]);
|
|
1504
|
-
return [];
|
|
1505
|
-
}
|
|
1506
|
-
const byStep = /* @__PURE__ */ new Map();
|
|
1507
|
-
for (const source of findings) {
|
|
1508
|
-
const steps = exactFindingSteps(trajectoryId, source);
|
|
1509
|
-
for (const step of steps) {
|
|
1510
|
-
if (byStep.has(step)) continue;
|
|
1511
|
-
byStep.set(step, makeFinding({
|
|
1512
|
-
analyst_id: analystId,
|
|
1513
|
-
area: "incorrect",
|
|
1514
|
-
subject: `incorrect-step-${step}`,
|
|
1515
|
-
claim: `Step ${step} is incorrect. ${source.claim}`,
|
|
1516
|
-
rationale: source.rationale,
|
|
1517
|
-
severity: source.severity,
|
|
1518
|
-
confidence: source.confidence,
|
|
1519
|
-
evidence_refs: [{
|
|
1520
|
-
kind: "span",
|
|
1521
|
-
uri: codeTraceStepEvidenceUri(trajectoryId, step),
|
|
1522
|
-
excerpt: source.evidence_refs.find((evidence) => evidence.uri === codeTraceStepEvidenceUri(trajectoryId, step))?.excerpt
|
|
1523
|
-
}],
|
|
1524
|
-
recommended_action: source.recommended_action,
|
|
1525
|
-
metadata: { sourceFindingId: source.finding_id },
|
|
1526
|
-
produced_at: source.produced_at,
|
|
1527
|
-
id_basis: `incorrect-step-${step}`
|
|
1528
|
-
}));
|
|
1529
|
-
}
|
|
1530
|
-
}
|
|
1531
|
-
return [...byStep].sort(([left], [right]) => left - right).map(([, finding]) => finding);
|
|
1532
|
-
}
|
|
1533
|
-
function exactFindingSteps(trajectoryId, finding) {
|
|
1534
|
-
if (finding.evidence_refs.length === 0) throw new Error(`model finding '${finding.finding_id}' has no step evidence`);
|
|
1535
|
-
const steps = finding.evidence_refs.map((evidence) => {
|
|
1536
|
-
const parsed = codeTraceStepFromEvidence(evidence.uri);
|
|
1537
|
-
if (!parsed || parsed.traceId !== trajectoryId) throw new Error(`model finding '${finding.finding_id}' cites non-case evidence '${evidence.uri}'`);
|
|
1538
|
-
return parsed.step;
|
|
1539
|
-
});
|
|
1540
|
-
return [...new Set(steps)];
|
|
1541
|
-
}
|
|
1542
|
-
//#endregion
|
|
1543
|
-
//#region src/analyst/benchmark-public-types.ts
|
|
1544
|
-
function requiredString(value, field) {
|
|
1545
|
-
const trimmed = value.trim();
|
|
1546
|
-
if (!trimmed) throw new TypeError(`${field} must be a non-empty string`);
|
|
1547
|
-
return trimmed;
|
|
1548
|
-
}
|
|
1549
|
-
function positiveSafeInteger(value, field) {
|
|
1550
|
-
if (!Number.isSafeInteger(value) || value < 1) throw new RangeError(`${field} must be a positive safe integer`);
|
|
1551
|
-
return value;
|
|
1552
|
-
}
|
|
1553
|
-
function safeInteger(value, field) {
|
|
1554
|
-
if (!Number.isSafeInteger(value)) throw new RangeError(`${field} must be a safe integer`);
|
|
1555
|
-
return value;
|
|
1556
|
-
}
|
|
1557
|
-
function isRecord(value) {
|
|
1558
|
-
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
1559
|
-
}
|
|
1560
|
-
//#endregion
|
|
1561
1488
|
//#region src/analyst/benchmark-verification-outcome.ts
|
|
1562
1489
|
const MAX_REPORTED_CHECKS = 20;
|
|
1563
1490
|
const SWE_MULTI_NO_TEST_RESULTS = "After applying the fix patch, no test results were captured when executing the test command.";
|
|
@@ -2157,6 +2084,391 @@ function isNodeError$1(error, code) {
|
|
|
2157
2084
|
return error instanceof Error && "code" in error && error.code === code;
|
|
2158
2085
|
}
|
|
2159
2086
|
//#endregion
|
|
2087
|
+
//#region src/analyst/benchmark-public-prompt.ts
|
|
2088
|
+
/** Widest contiguous failure block a model may report. The published corpus's
|
|
2089
|
+
* widest labeled block is 8 steps and its widest stage span is 9, so this bound
|
|
2090
|
+
* never binds honest enumeration; it caps how far one over-wide block can push
|
|
2091
|
+
* unlabeled steps into the precision denominator. */
|
|
2092
|
+
const MAX_INCORRECT_BLOCK_STEPS = 12;
|
|
2093
|
+
/** Most blocks a model may report for one trajectory. The published corpus's
|
|
2094
|
+
* densest case carries 4 disjoint labeled blocks. Together with the per-block
|
|
2095
|
+
* cap this bounds one case at 192 predicted steps without a second ceiling. */
|
|
2096
|
+
const MAX_INCORRECT_BLOCKS = 16;
|
|
2097
|
+
const TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS = [
|
|
2098
|
+
4096,
|
|
2099
|
+
2048,
|
|
2100
|
+
1024,
|
|
2101
|
+
512,
|
|
2102
|
+
256,
|
|
2103
|
+
128,
|
|
2104
|
+
64
|
|
2105
|
+
];
|
|
2106
|
+
const CODE_TRACE_BENCH_ANALYST_PROMPT = `Analyze exactly one coding-agent trajectory and its attached final verification.
|
|
2107
|
+
Your task is the CodeTraceBench incorrect-step task: identify every wrong state-changing action, bad hypothesis that drives an action, and regression.
|
|
2108
|
+
An incorrect step remains incorrect when the agent later recovers or the final verification passes; a trajectory that ends in success still contains every mistake the agent made along the way.
|
|
2109
|
+
Incorrect steps occur in contiguous failure blocks: one mistake plus every consecutive following step that commits to, compounds, or acts on it.
|
|
2110
|
+
Report each failure block as exactly one finding whose first_step is the block's first incorrect step and whose last_step is its last, covering every consecutive step between them.
|
|
2111
|
+
Set first_step to the first step that commits the mistake, not the step that planned it and not a later step that repeats it.
|
|
2112
|
+
Extend last_step one step at a time, and only while the next step independently satisfies the incorrect-step definition on its own action and its own following observation.
|
|
2113
|
+
Stop at the first step where the agent detects the problem, inspects it, or begins repairing it: a diagnostic probe, a test run that exposes the defect, or a repair action ends the block and is never inside it.
|
|
2114
|
+
A one-step block is a complete and correct answer.
|
|
2115
|
+
Every step inside a block is scored on its own: naming a correct step costs exactly as much as missing an incorrect one, and naming only the first step of a longer block forfeits every unnamed step.
|
|
2116
|
+
Report blocks separated by at least one correct step as separate findings, and never let two blocks overlap.
|
|
2117
|
+
Inspect the complete supplied trace data.
|
|
2118
|
+
Use the final-verification outcome as evidence about the final state, not as a rule for whether earlier steps were incorrect.
|
|
2119
|
+
For each candidate block, inspect every assistant action in it and its following observation.
|
|
2120
|
+
Admit a block only when you can point at the specific evidence it produced: name as consequence_step the step number whose action or observation shows the damage — a failing command, a wrong file state, a repeated failure, or rework the agent had to do because of this block. That step is the block's own last step when its observation already shows the damage, and a later step otherwise.
|
|
2121
|
+
When you cannot name that later step number from the trace you were given, drop the block; a plausible story about why a step looks wrong is not evidence that it was.
|
|
2122
|
+
Judge that consequence from the trajectory itself: a passing final verification is not evidence that a block caused nothing, and a failing final verification is not evidence that any particular block caused it.
|
|
2123
|
+
For every block, decide whether the agent escaped the failure.
|
|
2124
|
+
Mark escape_status "escaped" only when you can name the single later step that fully reversed the block, the agent needed no other step to recover, and nothing after that step revisits the same file, command, or hypothesis; write that step number in the rationale.
|
|
2125
|
+
Mark escape_status "unescaped" in every other case, including whenever you are unsure.
|
|
2126
|
+
A passing final verification never makes a block escaped: the agent may have made the mistake and repaired it over several steps, and those steps are still incorrect.
|
|
2127
|
+
Label a failed command when the assistant caused it through a wrong action or unsupported hypothesis.
|
|
2128
|
+
Label the later corrective action only when that action is itself wrong.
|
|
2129
|
+
Do not label a diagnostic probe merely because it exposes an earlier defect.
|
|
2130
|
+
Do not label a redundant but correct read or search; CodeTraceBench scores unuseful steps separately, and this run scores incorrect steps only.
|
|
2131
|
+
Do not label a step solely because final verification failed.
|
|
2132
|
+
When final verification is unavailable, use only directly observed trajectory evidence.
|
|
2133
|
+
Every step in a reported block MUST be the positive integer n from an existing assistant LLM span named step-<n>.
|
|
2134
|
+
Never select an EVALUATOR, TOOL, CHAIN, final-verification, benchmark-verification, or message-<n> span.
|
|
2135
|
+
Before emitting a finding, inspect every covered span's attributes.content and describe only the actions shown there.
|
|
2136
|
+
Report at most 16 blocks and at most 12 steps in one block; when more candidates than that exist, report the ones you can support with the clearest downstream evidence.
|
|
2137
|
+
When the trajectory has no incorrect steps, return an empty findings array.`;
|
|
2138
|
+
const AGENT_RX_PROMPT = `Analyze exactly one failed agent trajectory.
|
|
2139
|
+
Find the first unrecoverable critical failure, not every later symptom.
|
|
2140
|
+
Inspect the complete supplied trace data.
|
|
2141
|
+
Emit zero findings only when the trace does not contain enough evidence.
|
|
2142
|
+
Otherwise emit exactly one finding.
|
|
2143
|
+
Its category MUST be exactly one of:
|
|
2144
|
+
instruction-plan-adherence-failure
|
|
2145
|
+
invention-of-new-information
|
|
2146
|
+
invalid-invocation
|
|
2147
|
+
misinterpretation-of-tool-output-handoff-failure
|
|
2148
|
+
intent-plan-misalignment
|
|
2149
|
+
underspecified-user-intent
|
|
2150
|
+
intent-not-supported
|
|
2151
|
+
guardrails-triggered
|
|
2152
|
+
system-failure
|
|
2153
|
+
inconclusive
|
|
2154
|
+
Its step is the positive integer n from the first unrecoverable assistant span named step-<n>.`;
|
|
2155
|
+
const AGENT_RX_JSON_CONTRACT = `Each finding must contain only:
|
|
2156
|
+
- "step": a positive integer matching an existing assistant LLM span named step-<n>
|
|
2157
|
+
- "severity": "critical", "high", "medium", "low", or "info"
|
|
2158
|
+
- "claim": one sentence
|
|
2159
|
+
- "confidence": a number from 0 through 1
|
|
2160
|
+
- optional "rationale" and "recommended_action" strings
|
|
2161
|
+
- "category": one allowed failure category listed above`;
|
|
2162
|
+
const CODE_TRACE_JSON_CONTRACT = `Each finding is one contiguous failure block and must contain only:
|
|
2163
|
+
- "first_step": a positive integer, the block's first incorrect step, matching an existing assistant LLM span named step-<n>
|
|
2164
|
+
- "last_step": a positive integer >= first_step, the block's last incorrect step; every step from first_step through last_step must be an existing assistant LLM span, and a block spans at most 12 steps
|
|
2165
|
+
- "consequence_step": a positive integer >= first_step, the step whose action or following observation shows the damage this block caused; it may sit inside the block when the damage is already visible there
|
|
2166
|
+
- "escape_status": "escaped" only when one single later step fully reversed the block and nothing afterwards revisits it, "unescaped" otherwise and whenever you are unsure
|
|
2167
|
+
- "severity": "critical", "high", "medium", "low", or "info"
|
|
2168
|
+
- "claim": one sentence describing the block's failure
|
|
2169
|
+
- "confidence": a number from 0 through 1
|
|
2170
|
+
- optional "rationale" and "recommended_action" strings`;
|
|
2171
|
+
const AGENT_RX_RLM_CONTRACT = `Use the trace tools to inspect the action and its following observation.
|
|
2172
|
+
Emit exactly one finding whose subject is exactly one of the allowed failure categories.
|
|
2173
|
+
Cite exactly one assistant span named step-<n> as trace://<URL-encoded-trace-id>/span/step-<n>.
|
|
2174
|
+
The excerpt must quote the assistant action exactly.`;
|
|
2175
|
+
const CODE_TRACE_RLM_CONTRACT = `Use the trace tools rather than asking for the whole trajectory in the prompt.
|
|
2176
|
+
Keep retrieved trace objects in Python variables.
|
|
2177
|
+
Never print an entire trace, full source file, or more than 12000 characters in one iteration.
|
|
2178
|
+
Build a compact table of assistant step ids, actions, following observations, and final verification.
|
|
2179
|
+
Inspect suspicious steps with viewSpans or searchSpan instead of repeatedly printing the table.
|
|
2180
|
+
This runner emits no JSON fields, so the block is encoded in the finding's subject.
|
|
2181
|
+
Only findings_json is scored; your prose answer is ignored, so every incorrect block you identify must appear as a finding, never only in the answer.
|
|
2182
|
+
Emit exactly one finding per contiguous failure block.
|
|
2183
|
+
Set the finding's subject to incorrect-steps-<first_step>-<last_step>-<escape_status>-consequence-<consequence_step>, using the same four values the task defines; for a block covering only step 7 that the agent never escaped and whose damage shows at step 9, the subject is incorrect-steps-7-7-unescaped-consequence-9.
|
|
2184
|
+
The runner expands the block to one scored step per member and builds every scored citation itself.
|
|
2185
|
+
Cite the block's first step and its last step as trace://<URL-encoded-trace-id>/span/step-<n>, each excerpt an exact quote from that step's own action content.
|
|
2186
|
+
Give the rationale as the concrete downstream evidence visible at the consequence step.
|
|
2187
|
+
Submit as soon as every candidate failure block has a supported verdict.
|
|
2188
|
+
Return no finding for a clean trajectory.`;
|
|
2189
|
+
/** One-shot JSON transport prompt for the direct runner. */
|
|
2190
|
+
function publicBenchmarkSystemPrompt(dataset) {
|
|
2191
|
+
const fieldContract = dataset === "agentrx" ? AGENT_RX_JSON_CONTRACT : CODE_TRACE_JSON_CONTRACT;
|
|
2192
|
+
return `${publicBenchmarkTaskPrompt(dataset)}
|
|
2193
|
+
|
|
2194
|
+
${fieldContract}
|
|
2195
|
+
|
|
2196
|
+
Return exactly one JSON object with:
|
|
2197
|
+
- "report": a concise evidence-based explanation, at most 4000 characters
|
|
2198
|
+
- "findings": the strict finding array
|
|
2199
|
+
Use an empty findings array when the trace does not support a finding.
|
|
2200
|
+
Do not return a bare array, markdown, trace URIs, copied excerpts, or fields not listed above.
|
|
2201
|
+
The runner constructs exact trace URIs and action previews from each selected step.`;
|
|
2202
|
+
}
|
|
2203
|
+
/** Tool-loop prompt for the recursive runner. Same task, subject-encoded block. */
|
|
2204
|
+
function publicBenchmarkRlmInstructions(dataset) {
|
|
2205
|
+
const outputContract = dataset === "agentrx" ? AGENT_RX_RLM_CONTRACT : CODE_TRACE_RLM_CONTRACT;
|
|
2206
|
+
return `${publicBenchmarkTaskPrompt(dataset)}
|
|
2207
|
+
${outputContract}`;
|
|
2208
|
+
}
|
|
2209
|
+
function publicBenchmarkTaskPrompt(dataset) {
|
|
2210
|
+
return dataset === "agentrx" ? AGENT_RX_PROMPT : CODE_TRACE_BENCH_ANALYST_PROMPT;
|
|
2211
|
+
}
|
|
2212
|
+
/** Digest of every prompt a runner can send plus the shared transport limits.
|
|
2213
|
+
* Both runner contracts are hashed so an edit to either one changes the digest
|
|
2214
|
+
* a run records, whichever runner executed. */
|
|
2215
|
+
function publicBenchmarkProtocolSha256(dataset) {
|
|
2216
|
+
return sha256Digest(JSON.stringify({
|
|
2217
|
+
dataset,
|
|
2218
|
+
systemPrompt: publicBenchmarkSystemPrompt(dataset),
|
|
2219
|
+
rlmInstructions: publicBenchmarkRlmInstructions(dataset),
|
|
2220
|
+
transport: {
|
|
2221
|
+
attempts: 1,
|
|
2222
|
+
jsonMode: true,
|
|
2223
|
+
thinking: "disabled"
|
|
2224
|
+
},
|
|
2225
|
+
blockLimits: dataset === "agentrx" ? null : {
|
|
2226
|
+
maxBlocks: 16,
|
|
2227
|
+
maxBlockSteps: 12
|
|
2228
|
+
},
|
|
2229
|
+
traceProjectionAttributeByteCaps: TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS,
|
|
2230
|
+
evidence: {
|
|
2231
|
+
location: "model-selected-positive-integer-assistant-step",
|
|
2232
|
+
uri: "deterministic-trace-uri",
|
|
2233
|
+
excerpt: `exact-action-prefix-512`
|
|
2234
|
+
}
|
|
2235
|
+
}));
|
|
2236
|
+
}
|
|
2237
|
+
//#endregion
|
|
2238
|
+
//#region src/analyst/benchmark-public-adapters.ts
|
|
2239
|
+
function emptyPublicBenchmarkRunner() {
|
|
2240
|
+
return {
|
|
2241
|
+
id: "empty",
|
|
2242
|
+
analyze() {
|
|
2243
|
+
return {
|
|
2244
|
+
findings: [],
|
|
2245
|
+
usage: {
|
|
2246
|
+
calls: 0,
|
|
2247
|
+
tokens: {
|
|
2248
|
+
input: 0,
|
|
2249
|
+
output: 0
|
|
2250
|
+
},
|
|
2251
|
+
cost: {
|
|
2252
|
+
kind: "observed",
|
|
2253
|
+
usd: 0
|
|
2254
|
+
}
|
|
2255
|
+
},
|
|
2256
|
+
metadata: { baseline: "emit-no-findings" }
|
|
2257
|
+
};
|
|
2258
|
+
}
|
|
2259
|
+
};
|
|
2260
|
+
}
|
|
2261
|
+
async function adaptPublicBenchmarkFindings(options) {
|
|
2262
|
+
if (options.dataset === "agentrx") return {
|
|
2263
|
+
findings: adaptAgentRxFindings(options.trajectoryId, options.findings, options.analystId),
|
|
2264
|
+
diagnostics: void 0
|
|
2265
|
+
};
|
|
2266
|
+
return adaptCodeTraceFindings(options.trajectoryId, options.findings, options.analystId, options.store, options.signal);
|
|
2267
|
+
}
|
|
2268
|
+
function adaptAgentRxFindings(trajectoryId, findings, analystId) {
|
|
2269
|
+
if (findings.length === 0) return [];
|
|
2270
|
+
if (findings.length !== 1) throw new Error(`AgentRx model analyst must emit zero or one root cause, received ${findings.length}`);
|
|
2271
|
+
const source = findings[0];
|
|
2272
|
+
if (!source.subject) throw new Error("AgentRx model analyst finding is missing its failure-category subject");
|
|
2273
|
+
const steps = exactFindingSteps(trajectoryId, source);
|
|
2274
|
+
if (steps.length !== 1) throw new Error(`AgentRx model analyst must cite exactly one root-cause step, received ${steps.length}`);
|
|
2275
|
+
const [adapted] = agentRxPredictionsToFindings(trajectoryId, [{
|
|
2276
|
+
failure_case: source.subject,
|
|
2277
|
+
step_number: steps[0],
|
|
2278
|
+
description: source.rationale ?? source.claim
|
|
2279
|
+
}], {
|
|
2280
|
+
analystId,
|
|
2281
|
+
producedAt: source.produced_at,
|
|
2282
|
+
confidence: source.confidence
|
|
2283
|
+
});
|
|
2284
|
+
if (!adapted) throw new Error("AgentRx output adapter produced no root-cause finding");
|
|
2285
|
+
return [{
|
|
2286
|
+
...adapted,
|
|
2287
|
+
metadata: {
|
|
2288
|
+
...adapted.metadata,
|
|
2289
|
+
sourceFindingId: source.finding_id
|
|
2290
|
+
}
|
|
2291
|
+
}];
|
|
2292
|
+
}
|
|
2293
|
+
const CODE_TRACE_BLOCK_SUBJECT = /^incorrect-steps-(\d+)-(\d+)-(escaped|unescaped)-consequence-(\d+)$/;
|
|
2294
|
+
async function adaptCodeTraceFindings(trajectoryId, findings, analystId, store, signal) {
|
|
2295
|
+
const clean = findings.filter((finding) => finding.subject === "clean");
|
|
2296
|
+
if (clean.length > 0) {
|
|
2297
|
+
if (findings.length !== 1) throw new Error("CodeTraceBench model analyst mixed a clean verdict with incorrect steps");
|
|
2298
|
+
exactFindingSteps(trajectoryId, clean[0]);
|
|
2299
|
+
return {
|
|
2300
|
+
findings: [],
|
|
2301
|
+
diagnostics: emptyCodeTraceBlockDiagnostics()
|
|
2302
|
+
};
|
|
2303
|
+
}
|
|
2304
|
+
const blocks = [];
|
|
2305
|
+
const rejectedFindings = [];
|
|
2306
|
+
for (const source of findings) try {
|
|
2307
|
+
await validateCodeTraceFindingEvidence({
|
|
2308
|
+
trajectoryId,
|
|
2309
|
+
findings: [source],
|
|
2310
|
+
store,
|
|
2311
|
+
...signal ? { signal } : {}
|
|
2312
|
+
});
|
|
2313
|
+
blocks.push(codeTraceBlockFromFinding(trajectoryId, source));
|
|
2314
|
+
} catch (error) {
|
|
2315
|
+
rejectedFindings.push(`${source.finding_id}: ${error instanceof Error ? error.message : String(error)}`);
|
|
2316
|
+
}
|
|
2317
|
+
const expanded = await expandCodeTraceFailureBlocks({
|
|
2318
|
+
trajectoryId,
|
|
2319
|
+
blocks,
|
|
2320
|
+
store,
|
|
2321
|
+
analystId,
|
|
2322
|
+
...findings[0] ? { producedAt: findings[0].produced_at } : {},
|
|
2323
|
+
...signal ? { signal } : {}
|
|
2324
|
+
});
|
|
2325
|
+
return {
|
|
2326
|
+
findings: expanded.findings,
|
|
2327
|
+
diagnostics: {
|
|
2328
|
+
...expanded.diagnostics,
|
|
2329
|
+
rejectedFindings
|
|
2330
|
+
}
|
|
2331
|
+
};
|
|
2332
|
+
}
|
|
2333
|
+
function codeTraceBlockFromFinding(trajectoryId, source) {
|
|
2334
|
+
const parsed = CODE_TRACE_BLOCK_SUBJECT.exec(source.subject ?? "");
|
|
2335
|
+
if (!parsed) throw new Error(`CodeTraceBench model finding '${source.finding_id}' must set subject to incorrect-steps-<first>-<last>-<escaped|unescaped>-consequence-<step>, received '${source.subject ?? ""}'`);
|
|
2336
|
+
const firstStep = Number(parsed[1]);
|
|
2337
|
+
const lastStep = Number(parsed[2]);
|
|
2338
|
+
const consequenceStep = Number(parsed[4]);
|
|
2339
|
+
const cited = exactFindingSteps(trajectoryId, source);
|
|
2340
|
+
for (const step of cited) if (step < firstStep || step > lastStep) throw new Error(`CodeTraceBench model finding '${source.finding_id}' cites step ${step} outside its block ${firstStep}-${lastStep}`);
|
|
2341
|
+
return {
|
|
2342
|
+
firstStep,
|
|
2343
|
+
lastStep,
|
|
2344
|
+
consequenceStep,
|
|
2345
|
+
escapeStatus: parsed[3],
|
|
2346
|
+
severity: source.severity,
|
|
2347
|
+
claim: source.claim,
|
|
2348
|
+
confidence: source.confidence,
|
|
2349
|
+
...source.rationale === void 0 ? {} : { rationale: source.rationale },
|
|
2350
|
+
...source.recommended_action === void 0 ? {} : { recommendedAction: source.recommended_action },
|
|
2351
|
+
metadata: { sourceFindingId: source.finding_id }
|
|
2352
|
+
};
|
|
2353
|
+
}
|
|
2354
|
+
/**
|
|
2355
|
+
* Expand contiguous failure blocks into one scored finding per member step.
|
|
2356
|
+
*
|
|
2357
|
+
* The official scorer matches on area plus the exact step evidence URI, so
|
|
2358
|
+
* blocks never reach it: every runner reports blocks, and this function turns
|
|
2359
|
+
* them into the per-step findings the benchmark defines.
|
|
2360
|
+
*/
|
|
2361
|
+
async function expandCodeTraceFailureBlocks(options) {
|
|
2362
|
+
const diagnostics = emptyCodeTraceBlockDiagnostics();
|
|
2363
|
+
diagnostics.reportedBlocks = options.blocks.length;
|
|
2364
|
+
if (options.blocks.length === 0) return {
|
|
2365
|
+
findings: [],
|
|
2366
|
+
diagnostics
|
|
2367
|
+
};
|
|
2368
|
+
assertCodeTraceBlockShape(options.blocks);
|
|
2369
|
+
const boundarySteps = options.blocks.flatMap((block) => [block.firstStep, block.lastStep]);
|
|
2370
|
+
const derivedSteps = options.blocks.flatMap((block) => [block.consequenceStep, ...interiorSteps(block)]);
|
|
2371
|
+
const evidenceByStep = await resolveAssistantStepEvidence({
|
|
2372
|
+
trajectoryId: options.trajectoryId,
|
|
2373
|
+
steps: boundarySteps,
|
|
2374
|
+
optionalSteps: derivedSteps,
|
|
2375
|
+
store: options.store,
|
|
2376
|
+
...options.signal ? { signal: options.signal } : {}
|
|
2377
|
+
});
|
|
2378
|
+
const byStep = /* @__PURE__ */ new Map();
|
|
2379
|
+
for (const block of options.blocks) {
|
|
2380
|
+
if (!evidenceByStep.has(block.consequenceStep)) {
|
|
2381
|
+
diagnostics.blocksWithoutConsequenceEvidence.push(block);
|
|
2382
|
+
continue;
|
|
2383
|
+
}
|
|
2384
|
+
if (block.escapeStatus === "escaped") diagnostics.escapedBlocks += 1;
|
|
2385
|
+
for (let step = block.firstStep; step <= block.lastStep; step += 1) {
|
|
2386
|
+
if (!evidenceByStep.has(step)) {
|
|
2387
|
+
diagnostics.unresolvedBlockInteriorSteps.push(step);
|
|
2388
|
+
continue;
|
|
2389
|
+
}
|
|
2390
|
+
if (byStep.has(step)) {
|
|
2391
|
+
diagnostics.overlappingBlockSteps.push(step);
|
|
2392
|
+
continue;
|
|
2393
|
+
}
|
|
2394
|
+
byStep.set(step, block);
|
|
2395
|
+
}
|
|
2396
|
+
}
|
|
2397
|
+
return {
|
|
2398
|
+
findings: [...byStep].sort(([left], [right]) => left - right).map(([step, block]) => makeFinding({
|
|
2399
|
+
analyst_id: options.analystId,
|
|
2400
|
+
area: "incorrect",
|
|
2401
|
+
subject: `incorrect-step-${step}`,
|
|
2402
|
+
claim: `Step ${step} is incorrect. ${block.claim}`,
|
|
2403
|
+
rationale: block.rationale,
|
|
2404
|
+
severity: block.severity,
|
|
2405
|
+
confidence: block.confidence,
|
|
2406
|
+
evidence_refs: [evidenceByStep.get(step)],
|
|
2407
|
+
recommended_action: block.recommendedAction,
|
|
2408
|
+
metadata: {
|
|
2409
|
+
...block.metadata,
|
|
2410
|
+
block_first_step: block.firstStep,
|
|
2411
|
+
block_last_step: block.lastStep,
|
|
2412
|
+
block_consequence_step: block.consequenceStep,
|
|
2413
|
+
escape_status: block.escapeStatus
|
|
2414
|
+
},
|
|
2415
|
+
...options.producedAt === void 0 ? {} : { produced_at: options.producedAt },
|
|
2416
|
+
id_basis: `incorrect-step-${step}`
|
|
2417
|
+
})),
|
|
2418
|
+
diagnostics
|
|
2419
|
+
};
|
|
2420
|
+
}
|
|
2421
|
+
function assertCodeTraceBlockShape(blocks) {
|
|
2422
|
+
if (blocks.length > 16) throw new Error(`model reported ${blocks.length} failure blocks; the maximum is 16`);
|
|
2423
|
+
for (const block of blocks) {
|
|
2424
|
+
if (block.lastStep < block.firstStep) throw new Error(`failure block last_step ${block.lastStep} precedes first_step ${block.firstStep}`);
|
|
2425
|
+
const length = block.lastStep - block.firstStep + 1;
|
|
2426
|
+
if (length > 12) throw new Error(`failure block spans ${length} steps; the maximum is 12`);
|
|
2427
|
+
if (block.consequenceStep < block.firstStep) throw new Error(`failure block consequence_step ${block.consequenceStep} precedes first_step ${block.firstStep}`);
|
|
2428
|
+
}
|
|
2429
|
+
}
|
|
2430
|
+
function interiorSteps(block) {
|
|
2431
|
+
const steps = [];
|
|
2432
|
+
for (let step = block.firstStep + 1; step < block.lastStep; step += 1) steps.push(step);
|
|
2433
|
+
return steps;
|
|
2434
|
+
}
|
|
2435
|
+
function emptyCodeTraceBlockDiagnostics() {
|
|
2436
|
+
return {
|
|
2437
|
+
reportedBlocks: 0,
|
|
2438
|
+
escapedBlocks: 0,
|
|
2439
|
+
blocksWithoutConsequenceEvidence: [],
|
|
2440
|
+
unresolvedBlockInteriorSteps: [],
|
|
2441
|
+
overlappingBlockSteps: []
|
|
2442
|
+
};
|
|
2443
|
+
}
|
|
2444
|
+
function exactFindingSteps(trajectoryId, finding) {
|
|
2445
|
+
if (finding.evidence_refs.length === 0) throw new Error(`model finding '${finding.finding_id}' has no step evidence`);
|
|
2446
|
+
const steps = finding.evidence_refs.map((evidence) => {
|
|
2447
|
+
const parsed = codeTraceStepFromEvidence(evidence.uri);
|
|
2448
|
+
if (!parsed || parsed.traceId !== trajectoryId) throw new Error(`model finding '${finding.finding_id}' cites non-case evidence '${evidence.uri}'`);
|
|
2449
|
+
return parsed.step;
|
|
2450
|
+
});
|
|
2451
|
+
return [...new Set(steps)];
|
|
2452
|
+
}
|
|
2453
|
+
//#endregion
|
|
2454
|
+
//#region src/analyst/benchmark-public-types.ts
|
|
2455
|
+
function requiredString(value, field) {
|
|
2456
|
+
const trimmed = value.trim();
|
|
2457
|
+
if (!trimmed) throw new TypeError(`${field} must be a non-empty string`);
|
|
2458
|
+
return trimmed;
|
|
2459
|
+
}
|
|
2460
|
+
function positiveSafeInteger(value, field) {
|
|
2461
|
+
if (!Number.isSafeInteger(value) || value < 1) throw new RangeError(`${field} must be a positive safe integer`);
|
|
2462
|
+
return value;
|
|
2463
|
+
}
|
|
2464
|
+
function safeInteger(value, field) {
|
|
2465
|
+
if (!Number.isSafeInteger(value)) throw new RangeError(`${field} must be a safe integer`);
|
|
2466
|
+
return value;
|
|
2467
|
+
}
|
|
2468
|
+
function isRecord(value) {
|
|
2469
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
2470
|
+
}
|
|
2471
|
+
//#endregion
|
|
2160
2472
|
//#region src/analyst/benchmark-public-data.ts
|
|
2161
2473
|
const DEFAULT_MAX_PUBLIC_BENCHMARK_LABEL_BYTES = 256 * 1024 * 1024;
|
|
2162
2474
|
const INPUT_OPEN_FLAGS = constants.O_RDONLY | (constants.O_NOFOLLOW ?? 0);
|
|
@@ -2655,15 +2967,6 @@ function fileContext() {
|
|
|
2655
2967
|
}
|
|
2656
2968
|
//#endregion
|
|
2657
2969
|
//#region src/analyst/benchmark-public-model.ts
|
|
2658
|
-
const TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS = [
|
|
2659
|
-
4096,
|
|
2660
|
-
2048,
|
|
2661
|
-
1024,
|
|
2662
|
-
512,
|
|
2663
|
-
256,
|
|
2664
|
-
128,
|
|
2665
|
-
64
|
|
2666
|
-
];
|
|
2667
2970
|
/** One-shot JSON baseline. This is not a recursive trace analyst. */
|
|
2668
2971
|
function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
2669
2972
|
const model = requiredString(config.model, "model");
|
|
@@ -2677,7 +2980,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
2677
2980
|
responseCacheDir: requiredString(config.durability.responseCacheDir, "durability.responseCacheDir")
|
|
2678
2981
|
} : void 0;
|
|
2679
2982
|
const actor = dataset === "agentrx" ? "agentrx-root-cause-localizer" : "codetracebench-step-localizer";
|
|
2680
|
-
const outputAdapter = dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-
|
|
2983
|
+
const outputAdapter = dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-block";
|
|
2681
2984
|
const llmOptions = {
|
|
2682
2985
|
baseUrl,
|
|
2683
2986
|
apiKey,
|
|
@@ -2697,6 +3000,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
2697
3000
|
benchmarkRepetition: String(context.repetition)
|
|
2698
3001
|
};
|
|
2699
3002
|
let rawPredictions = [];
|
|
3003
|
+
let rejectedBlocks = [];
|
|
2700
3004
|
let modelFindings = [];
|
|
2701
3005
|
let providerModel = model;
|
|
2702
3006
|
let producedAt;
|
|
@@ -2748,6 +3052,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
2748
3052
|
};
|
|
2749
3053
|
const response = parsePublicBenchmarkModelResponse(dataset, cached.response);
|
|
2750
3054
|
rawPredictions = response.findings;
|
|
3055
|
+
rejectedBlocks = response.rejectedBlocks;
|
|
2751
3056
|
providerModel = cached.metadata.providerModel;
|
|
2752
3057
|
producedAt = cached.metadata.producedAt;
|
|
2753
3058
|
modelMetadata = {
|
|
@@ -2784,7 +3089,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
2784
3089
|
...cacheIdentity,
|
|
2785
3090
|
callId: providerCallId,
|
|
2786
3091
|
status: "succeeded",
|
|
2787
|
-
response,
|
|
3092
|
+
response: completed.value,
|
|
2788
3093
|
metadata: {
|
|
2789
3094
|
providerModel: completed.result.model,
|
|
2790
3095
|
providerDurationMs: completed.result.durationMs,
|
|
@@ -2816,6 +3121,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
2816
3121
|
if (!paid.succeeded) throw paid.error;
|
|
2817
3122
|
const response = paid.value.response;
|
|
2818
3123
|
rawPredictions = response.findings;
|
|
3124
|
+
rejectedBlocks = response.rejectedBlocks;
|
|
2819
3125
|
providerModel = paid.value.result.model;
|
|
2820
3126
|
producedAt = paid.value.producedAt;
|
|
2821
3127
|
modelMetadata = {
|
|
@@ -2828,7 +3134,7 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
2828
3134
|
cost: costReceiptMetadata(paid.receipt)
|
|
2829
3135
|
};
|
|
2830
3136
|
}
|
|
2831
|
-
|
|
3137
|
+
const converted = await publicBenchmarkPredictionsToFindings({
|
|
2832
3138
|
dataset,
|
|
2833
3139
|
trajectoryId,
|
|
2834
3140
|
predictions: rawPredictions,
|
|
@@ -2838,6 +3144,14 @@ function createPublicBenchmarkDirectRunner(dataset, config) {
|
|
|
2838
3144
|
producedAt: requiredString(producedAt ?? "", "finding producedAt"),
|
|
2839
3145
|
...context.signal ? { signal: context.signal } : {}
|
|
2840
3146
|
});
|
|
3147
|
+
modelFindings = converted.findings;
|
|
3148
|
+
if (converted.diagnostics) modelMetadata = {
|
|
3149
|
+
...modelMetadata,
|
|
3150
|
+
blockDiagnostics: {
|
|
3151
|
+
...converted.diagnostics,
|
|
3152
|
+
rejectedBlocks
|
|
3153
|
+
}
|
|
3154
|
+
};
|
|
2841
3155
|
if (dataset === "codetracebench") await validateCodeTraceFindingEvidence({
|
|
2842
3156
|
trajectoryId,
|
|
2843
3157
|
findings: modelFindings,
|
|
@@ -2935,7 +3249,7 @@ const ModelSeveritySchema = z.enum([
|
|
|
2935
3249
|
"low",
|
|
2936
3250
|
"info"
|
|
2937
3251
|
]);
|
|
2938
|
-
const
|
|
3252
|
+
const AgentRxPredictionSchema = z.object({
|
|
2939
3253
|
step: z.number().int().positive(),
|
|
2940
3254
|
severity: ModelSeveritySchema,
|
|
2941
3255
|
claim: z.string().min(1),
|
|
@@ -2943,6 +3257,34 @@ const PublicBenchmarkPredictionSchema = z.object({
|
|
|
2943
3257
|
rationale: z.string().min(1).optional(),
|
|
2944
3258
|
recommended_action: z.string().min(1).optional()
|
|
2945
3259
|
}).strict();
|
|
3260
|
+
const CodeTraceBlockPredictionSchema = z.object({
|
|
3261
|
+
first_step: z.number().int().positive(),
|
|
3262
|
+
last_step: z.number().int().positive(),
|
|
3263
|
+
consequence_step: z.number().int().positive(),
|
|
3264
|
+
escape_status: z.enum(["escaped", "unescaped"]),
|
|
3265
|
+
severity: ModelSeveritySchema,
|
|
3266
|
+
claim: z.string().min(1),
|
|
3267
|
+
confidence: z.number().min(0).max(1),
|
|
3268
|
+
rationale: z.string().min(1).optional(),
|
|
3269
|
+
recommended_action: z.string().min(1).optional()
|
|
3270
|
+
}).strict().superRefine((block, ctx) => {
|
|
3271
|
+
if (block.last_step < block.first_step) {
|
|
3272
|
+
ctx.addIssue({
|
|
3273
|
+
code: z.ZodIssueCode.custom,
|
|
3274
|
+
message: `failure block last_step ${block.last_step} precedes first_step ${block.first_step}`
|
|
3275
|
+
});
|
|
3276
|
+
return;
|
|
3277
|
+
}
|
|
3278
|
+
const length = block.last_step - block.first_step + 1;
|
|
3279
|
+
if (length > 12) ctx.addIssue({
|
|
3280
|
+
code: z.ZodIssueCode.custom,
|
|
3281
|
+
message: `failure block spans ${length} steps; the maximum is 12`
|
|
3282
|
+
});
|
|
3283
|
+
if (block.consequence_step < block.first_step) ctx.addIssue({
|
|
3284
|
+
code: z.ZodIssueCode.custom,
|
|
3285
|
+
message: `failure block consequence_step ${block.consequence_step} precedes first_step ${block.first_step}`
|
|
3286
|
+
});
|
|
3287
|
+
});
|
|
2946
3288
|
const AgentRxCategorySchema = z.enum([
|
|
2947
3289
|
"instruction-plan-adherence-failure",
|
|
2948
3290
|
"invention-of-new-information",
|
|
@@ -2955,28 +3297,52 @@ const AgentRxCategorySchema = z.enum([
|
|
|
2955
3297
|
"system-failure",
|
|
2956
3298
|
"inconclusive"
|
|
2957
3299
|
]);
|
|
2958
|
-
const
|
|
3300
|
+
const CodeTraceModelResponseEnvelopeSchema = z.object({
|
|
2959
3301
|
report: z.string().min(1).max(4e3),
|
|
2960
|
-
findings: z.array(
|
|
3302
|
+
findings: z.array(z.unknown()).max(16)
|
|
2961
3303
|
}).strict();
|
|
2962
3304
|
const AgentRxModelResponseSchema = z.object({
|
|
2963
3305
|
report: z.string().min(1).max(4e3),
|
|
2964
|
-
findings: z.array(
|
|
3306
|
+
findings: z.array(AgentRxPredictionSchema.extend({ category: AgentRxCategorySchema }).strict()).max(1)
|
|
2965
3307
|
}).strict();
|
|
2966
3308
|
function parsePublicBenchmarkModelResponse(dataset, value) {
|
|
2967
|
-
|
|
3309
|
+
if (dataset === "agentrx") return {
|
|
3310
|
+
...AgentRxModelResponseSchema.parse(value),
|
|
3311
|
+
rejectedBlocks: []
|
|
3312
|
+
};
|
|
3313
|
+
const envelope = CodeTraceModelResponseEnvelopeSchema.parse(value);
|
|
3314
|
+
const findings = [];
|
|
3315
|
+
const rejectedBlocks = [];
|
|
3316
|
+
for (const [index, block] of envelope.findings.entries()) {
|
|
3317
|
+
const parsed = CodeTraceBlockPredictionSchema.safeParse(block);
|
|
3318
|
+
if (parsed.success) {
|
|
3319
|
+
findings.push(parsed.data);
|
|
3320
|
+
continue;
|
|
3321
|
+
}
|
|
3322
|
+
rejectedBlocks.push(`block ${index}: ${parsed.error.issues.map((issue) => `${issue.path.join(".") || "<root>"} ${issue.message}`).join("; ")}`);
|
|
3323
|
+
}
|
|
3324
|
+
if (findings.length === 0 && envelope.findings.length > 0) throw new ValidationError(`every reported failure block was malformed: ${rejectedBlocks.join(" | ")}`);
|
|
3325
|
+
return {
|
|
3326
|
+
report: envelope.report,
|
|
3327
|
+
findings,
|
|
3328
|
+
rejectedBlocks
|
|
3329
|
+
};
|
|
2968
3330
|
}
|
|
2969
3331
|
async function publicBenchmarkPredictionsToFindings(options) {
|
|
2970
|
-
if (options.predictions.length === 0) return
|
|
2971
|
-
|
|
2972
|
-
|
|
2973
|
-
|
|
2974
|
-
store: options.store,
|
|
2975
|
-
...options.signal ? { signal: options.signal } : {}
|
|
2976
|
-
});
|
|
3332
|
+
if (options.predictions.length === 0 && options.dataset === "agentrx") return {
|
|
3333
|
+
findings: [],
|
|
3334
|
+
diagnostics: void 0
|
|
3335
|
+
};
|
|
2977
3336
|
if (options.dataset === "agentrx") {
|
|
2978
3337
|
const prediction = options.predictions[0];
|
|
3338
|
+
if (!("step" in prediction)) throw new Error("AgentRx model output must name a single root-cause step");
|
|
2979
3339
|
if (!prediction.category) throw new Error("AgentRx model output is missing its failure category");
|
|
3340
|
+
const evidenceByStep = await resolveAssistantStepEvidence({
|
|
3341
|
+
trajectoryId: options.trajectoryId,
|
|
3342
|
+
steps: [prediction.step],
|
|
3343
|
+
store: options.store,
|
|
3344
|
+
...options.signal ? { signal: options.signal } : {}
|
|
3345
|
+
});
|
|
2980
3346
|
const [finding] = agentRxPredictionsToFindings(options.trajectoryId, [{
|
|
2981
3347
|
failure_case: prediction.category,
|
|
2982
3348
|
step_number: prediction.step,
|
|
@@ -2987,69 +3353,44 @@ async function publicBenchmarkPredictionsToFindings(options) {
|
|
|
2987
3353
|
confidence: prediction.confidence
|
|
2988
3354
|
});
|
|
2989
3355
|
if (!finding) throw new Error("AgentRx output adapter produced no root-cause finding");
|
|
2990
|
-
return
|
|
2991
|
-
|
|
2992
|
-
|
|
3356
|
+
return {
|
|
3357
|
+
findings: [{
|
|
3358
|
+
...finding,
|
|
3359
|
+
evidence_refs: [evidenceByStep.get(prediction.step)],
|
|
3360
|
+
metadata: {
|
|
3361
|
+
...finding.metadata,
|
|
3362
|
+
model: options.providerModel
|
|
3363
|
+
}
|
|
3364
|
+
}],
|
|
3365
|
+
diagnostics: void 0
|
|
3366
|
+
};
|
|
3367
|
+
}
|
|
3368
|
+
const blocks = options.predictions.map((prediction) => {
|
|
3369
|
+
if (!("first_step" in prediction)) throw new Error("CodeTraceBench model output must report first_step/last_step failure blocks");
|
|
3370
|
+
return {
|
|
3371
|
+
firstStep: prediction.first_step,
|
|
3372
|
+
lastStep: prediction.last_step,
|
|
3373
|
+
consequenceStep: prediction.consequence_step,
|
|
3374
|
+
escapeStatus: prediction.escape_status,
|
|
3375
|
+
severity: prediction.severity,
|
|
3376
|
+
claim: prediction.claim,
|
|
3377
|
+
confidence: prediction.confidence,
|
|
3378
|
+
...prediction.rationale === void 0 ? {} : { rationale: prediction.rationale },
|
|
3379
|
+
...prediction.recommended_action === void 0 ? {} : { recommendedAction: prediction.recommended_action },
|
|
2993
3380
|
metadata: {
|
|
2994
|
-
|
|
3381
|
+
analysis_mode: "direct-baseline",
|
|
2995
3382
|
model: options.providerModel
|
|
2996
3383
|
}
|
|
2997
|
-
}
|
|
2998
|
-
}
|
|
2999
|
-
|
|
3000
|
-
|
|
3001
|
-
|
|
3002
|
-
|
|
3003
|
-
|
|
3004
|
-
|
|
3005
|
-
|
|
3006
|
-
|
|
3007
|
-
severity: prediction.severity,
|
|
3008
|
-
confidence: prediction.confidence,
|
|
3009
|
-
evidence_refs: [evidenceByStep.get(step)],
|
|
3010
|
-
recommended_action: prediction.recommended_action,
|
|
3011
|
-
metadata: {
|
|
3012
|
-
analysis_mode: "direct-baseline",
|
|
3013
|
-
model: options.providerModel
|
|
3014
|
-
},
|
|
3015
|
-
produced_at: options.producedAt,
|
|
3016
|
-
id_basis: `incorrect-step-${step}`
|
|
3017
|
-
}));
|
|
3018
|
-
}
|
|
3019
|
-
function publicBenchmarkSystemPrompt(dataset) {
|
|
3020
|
-
return `${dataset === "agentrx" ? AGENT_RX_PROMPT : CODE_TRACE_BENCH_ANALYST_PROMPT}
|
|
3021
|
-
|
|
3022
|
-
Each finding must contain only:
|
|
3023
|
-
- "step": a positive integer matching an existing assistant LLM span named step-<n>
|
|
3024
|
-
- "severity": "critical", "high", "medium", "low", or "info"
|
|
3025
|
-
- "claim": one sentence
|
|
3026
|
-
- "confidence": a number from 0 through 1
|
|
3027
|
-
- optional "rationale" and "recommended_action" strings
|
|
3028
|
-
${dataset === "agentrx" ? `- "category": one allowed failure category listed above` : ""}
|
|
3029
|
-
|
|
3030
|
-
Return exactly one JSON object with:
|
|
3031
|
-
- "report": a concise evidence-based explanation, at most 4000 characters
|
|
3032
|
-
- "findings": the strict finding array
|
|
3033
|
-
Use an empty findings array when the trace does not support a finding.
|
|
3034
|
-
Do not return a bare array, markdown, trace URIs, copied excerpts, or fields not listed above.
|
|
3035
|
-
The runner constructs exact trace URIs and action previews from each selected step.`;
|
|
3036
|
-
}
|
|
3037
|
-
function publicBenchmarkProtocolSha256(dataset) {
|
|
3038
|
-
return sha256Digest(JSON.stringify({
|
|
3039
|
-
dataset,
|
|
3040
|
-
systemPrompt: publicBenchmarkSystemPrompt(dataset),
|
|
3041
|
-
transport: {
|
|
3042
|
-
attempts: 1,
|
|
3043
|
-
jsonMode: true,
|
|
3044
|
-
thinking: "disabled"
|
|
3045
|
-
},
|
|
3046
|
-
traceProjectionAttributeByteCaps: TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS,
|
|
3047
|
-
evidence: {
|
|
3048
|
-
location: "model-selected-positive-integer-assistant-step",
|
|
3049
|
-
uri: "deterministic-trace-uri",
|
|
3050
|
-
excerpt: `exact-action-prefix-512`
|
|
3051
|
-
}
|
|
3052
|
-
}));
|
|
3384
|
+
};
|
|
3385
|
+
});
|
|
3386
|
+
return expandCodeTraceFailureBlocks({
|
|
3387
|
+
trajectoryId: options.trajectoryId,
|
|
3388
|
+
blocks,
|
|
3389
|
+
store: options.store,
|
|
3390
|
+
analystId: options.analystId,
|
|
3391
|
+
producedAt: options.producedAt,
|
|
3392
|
+
...options.signal ? { signal: options.signal } : {}
|
|
3393
|
+
});
|
|
3053
3394
|
}
|
|
3054
3395
|
async function prepareSingleTraceContext(store, context) {
|
|
3055
3396
|
const storeContext = context.signal ? { signal: context.signal } : void 0;
|
|
@@ -3069,40 +3410,6 @@ async function prepareSingleTraceContext(store, context) {
|
|
|
3069
3410
|
});
|
|
3070
3411
|
}
|
|
3071
3412
|
}
|
|
3072
|
-
const AGENT_RX_PROMPT = `Analyze exactly one failed agent trajectory.
|
|
3073
|
-
Find the first unrecoverable critical failure, not every later symptom.
|
|
3074
|
-
Inspect the complete supplied trace data.
|
|
3075
|
-
Emit zero findings only when the trace does not contain enough evidence.
|
|
3076
|
-
Otherwise emit exactly one finding.
|
|
3077
|
-
Its category MUST be exactly one of:
|
|
3078
|
-
instruction-plan-adherence-failure
|
|
3079
|
-
invention-of-new-information
|
|
3080
|
-
invalid-invocation
|
|
3081
|
-
misinterpretation-of-tool-output-handoff-failure
|
|
3082
|
-
intent-plan-misalignment
|
|
3083
|
-
underspecified-user-intent
|
|
3084
|
-
intent-not-supported
|
|
3085
|
-
guardrails-triggered
|
|
3086
|
-
system-failure
|
|
3087
|
-
inconclusive
|
|
3088
|
-
Its step is the positive integer n from the first unrecoverable assistant span named step-<n>.`;
|
|
3089
|
-
const CODE_TRACE_BENCH_ANALYST_PROMPT = `Analyze exactly one coding-agent trajectory and its attached final verification.
|
|
3090
|
-
Your task is the CodeTraceBench incorrect-step task: identify every wrong state-changing action, bad hypothesis that drives an action, and regression.
|
|
3091
|
-
An incorrect step remains incorrect when the agent later recovers or the final verification passes.
|
|
3092
|
-
Inspect the complete supplied trace data.
|
|
3093
|
-
Use the final-verification outcome as evidence about the final state, not as a rule for whether earlier steps were incorrect.
|
|
3094
|
-
For each candidate, inspect the assistant action and its following observation.
|
|
3095
|
-
Label a failed command when the assistant caused it through a wrong action or unsupported hypothesis.
|
|
3096
|
-
Label the later corrective action only when that action is itself wrong.
|
|
3097
|
-
Do not label a diagnostic probe merely because it exposes an earlier defect.
|
|
3098
|
-
Do not label a redundant but correct read or search; CodeTraceBench scores unuseful steps separately, and this run scores incorrect steps only.
|
|
3099
|
-
Do not label a step solely because final verification failed.
|
|
3100
|
-
When final verification is unavailable, use only directly observed trajectory evidence.
|
|
3101
|
-
Emit one finding per incorrect assistant step.
|
|
3102
|
-
Each finding's step MUST be the positive integer n from an existing assistant LLM span named step-<n>.
|
|
3103
|
-
Never select an EVALUATOR, TOOL, CHAIN, final-verification, benchmark-verification, or message-<n> span.
|
|
3104
|
-
Before emitting a finding, inspect its candidate span's attributes.content and describe only the action shown there.
|
|
3105
|
-
When the trajectory has no incorrect steps, return an empty findings array.`;
|
|
3106
3413
|
function trajectoryIdFromCaseId$1(dataset, caseId) {
|
|
3107
3414
|
const prefix = dataset === "agentrx" ? "agentrx:" : "codetrace:";
|
|
3108
3415
|
if (!caseId.startsWith(prefix) || caseId.length === prefix.length) throw new Error(`unexpected ${dataset} benchmark case id '${caseId}'`);
|
|
@@ -3110,41 +3417,13 @@ function trajectoryIdFromCaseId$1(dataset, caseId) {
|
|
|
3110
3417
|
}
|
|
3111
3418
|
//#endregion
|
|
3112
3419
|
//#region src/analyst/benchmark-public-rlm.ts
|
|
3113
|
-
const AGENT_RX_RLM_INSTRUCTIONS = `Analyze exactly one failed agent trajectory.
|
|
3114
|
-
Find the first unrecoverable critical failure, not every later symptom.
|
|
3115
|
-
Use the trace tools to inspect the action and its following observation.
|
|
3116
|
-
Emit zero findings only when evidence does not support a root cause.
|
|
3117
|
-
Otherwise emit exactly one finding whose subject is exactly one of:
|
|
3118
|
-
instruction-plan-adherence-failure
|
|
3119
|
-
invention-of-new-information
|
|
3120
|
-
invalid-invocation
|
|
3121
|
-
misinterpretation-of-tool-output-handoff-failure
|
|
3122
|
-
intent-plan-misalignment
|
|
3123
|
-
underspecified-user-intent
|
|
3124
|
-
intent-not-supported
|
|
3125
|
-
guardrails-triggered
|
|
3126
|
-
system-failure
|
|
3127
|
-
inconclusive
|
|
3128
|
-
Cite exactly one assistant span named step-<n> as trace://<trace-id>/span/step-<n>.
|
|
3129
|
-
The excerpt must quote the assistant action exactly.`;
|
|
3130
|
-
const CODE_TRACE_RLM_INSTRUCTIONS = `${CODE_TRACE_BENCH_ANALYST_PROMPT}
|
|
3131
|
-
Use the trace tools rather than asking for the whole trajectory in the prompt.
|
|
3132
|
-
Keep retrieved trace objects in Python variables.
|
|
3133
|
-
Never print an entire trace, full source file, or more than 12000 characters in one iteration.
|
|
3134
|
-
Build a compact table of assistant step ids, actions, following observations, and final verification.
|
|
3135
|
-
Inspect suspicious steps with viewSpans or searchSpan instead of repeatedly printing the table.
|
|
3136
|
-
Submit as soon as every state-changing assistant step has a supported verdict.
|
|
3137
|
-
For each incorrect step, set subject to incorrect-step-<n>.
|
|
3138
|
-
Cite the assistant span as trace://<URL-encoded-trace-id>/span/step-<n>.
|
|
3139
|
-
The evidence excerpt must be an exact quote from that span's action content.
|
|
3140
|
-
Return no finding for a clean trajectory.`;
|
|
3141
3420
|
/** Public benchmark candidate that runs the actual recursive trace analyst. */
|
|
3142
3421
|
function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
3143
3422
|
const costLedger = config.costLedger ?? new CostLedger();
|
|
3144
3423
|
const limits = {
|
|
3145
|
-
maxIterations: config.dspyRlm?.maxIterations ??
|
|
3146
|
-
maxLlmCalls: config.dspyRlm?.maxLlmCalls ??
|
|
3147
|
-
maxToolCalls: config.dspyRlm?.maxToolCalls ??
|
|
3424
|
+
maxIterations: config.dspyRlm?.maxIterations ?? 14,
|
|
3425
|
+
maxLlmCalls: config.dspyRlm?.maxLlmCalls ?? 8,
|
|
3426
|
+
maxToolCalls: config.dspyRlm?.maxToolCalls ?? 80,
|
|
3148
3427
|
maxOutputChars: config.dspyRlm?.maxOutputChars ?? 8e3
|
|
3149
3428
|
};
|
|
3150
3429
|
const pricing = config.pricing ?? pricingForModel(config.model);
|
|
@@ -3205,20 +3484,22 @@ function createPublicBenchmarkRlmRunner(dataset, config) {
|
|
|
3205
3484
|
},
|
|
3206
3485
|
produced_at: producedAt
|
|
3207
3486
|
}));
|
|
3208
|
-
const
|
|
3209
|
-
|
|
3487
|
+
const adapted = await adaptPublicBenchmarkFindings({
|
|
3488
|
+
dataset,
|
|
3210
3489
|
trajectoryId,
|
|
3211
|
-
findings,
|
|
3490
|
+
findings: rawFindings,
|
|
3491
|
+
analystId: "dspy-rlm",
|
|
3212
3492
|
store: input.traceStore,
|
|
3213
3493
|
...context.signal ? { signal: context.signal } : {}
|
|
3214
3494
|
});
|
|
3215
3495
|
return {
|
|
3216
|
-
findings,
|
|
3496
|
+
findings: adapted.findings,
|
|
3217
3497
|
usage,
|
|
3218
3498
|
metadata: {
|
|
3219
3499
|
analysisMode: "recursive",
|
|
3220
3500
|
engine: "dspy-rlm",
|
|
3221
3501
|
protocolSha256: publicBenchmarkProtocolSha256(dataset),
|
|
3502
|
+
...adapted.diagnostics ? { blockDiagnostics: adapted.diagnostics } : {},
|
|
3222
3503
|
answer: completed.answer,
|
|
3223
3504
|
trajectory: completed.trajectory,
|
|
3224
3505
|
modelCalls: completed.modelCalls,
|
|
@@ -3250,7 +3531,7 @@ function publicBenchmarkDefinition(dataset, limits) {
|
|
|
3250
3531
|
area: dataset === "agentrx" ? "root-cause" : "incorrect",
|
|
3251
3532
|
version: "1.0.0",
|
|
3252
3533
|
question: dataset === "agentrx" ? "What is the first unrecoverable root cause in this failed trajectory?" : "Which assistant steps are incorrect under the CodeTraceBench definition?",
|
|
3253
|
-
instructions: dataset
|
|
3534
|
+
instructions: publicBenchmarkRlmInstructions(dataset),
|
|
3254
3535
|
toolGroup: "singleTrace",
|
|
3255
3536
|
limits
|
|
3256
3537
|
};
|
|
@@ -3341,7 +3622,7 @@ function createRunIdentity(config, prepared) {
|
|
|
3341
3622
|
analystProtocolSha256: publicBenchmarkProtocolSha256(config.dataset),
|
|
3342
3623
|
implementationSha256: ANALYST_BENCHMARK_IMPLEMENTATION_SHA256,
|
|
3343
3624
|
dependencyLockSha256: ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256,
|
|
3344
|
-
runnerIds: ["empty",
|
|
3625
|
+
runnerIds: ["empty", config.analyst]
|
|
3345
3626
|
},
|
|
3346
3627
|
inputs: {
|
|
3347
3628
|
labelsSha256: prepared.labelsSha256,
|
|
@@ -3512,7 +3793,7 @@ function createObservationAppender(path, runIdentitySha256, progress) {
|
|
|
3512
3793
|
return write;
|
|
3513
3794
|
};
|
|
3514
3795
|
}
|
|
3515
|
-
async function readProgress(path, runIdentitySha256, caseIds, repetitions) {
|
|
3796
|
+
async function readProgress(path, runIdentitySha256, caseIds, repetitions, analystRunnerId) {
|
|
3516
3797
|
const rawLines = (await readRegularFile(path, "benchmark observation log")).split("\n");
|
|
3517
3798
|
if (rawLines.at(-1) === "") rawLines.pop();
|
|
3518
3799
|
const observations = [];
|
|
@@ -3546,7 +3827,7 @@ async function readProgress(path, runIdentitySha256, caseIds, repetitions) {
|
|
|
3546
3827
|
previousRowSha256: parsed.previousRowSha256,
|
|
3547
3828
|
observation
|
|
3548
3829
|
}) !== parsed.rowSha256) throw new Error(`benchmark observation row ${index + 1} digest does not match its contents`);
|
|
3549
|
-
if (!allowedCases.has(observation.caseId) || observation.runnerId !== "empty" && observation.runnerId !==
|
|
3830
|
+
if (!allowedCases.has(observation.caseId) || observation.runnerId !== "empty" && observation.runnerId !== analystRunnerId || observation.repetition >= repetitions || observation.executionIndex >= plannedObservationCount) throw new Error(`benchmark observation row ${index + 1} does not match a planned case, runner, and repetition`);
|
|
3550
3831
|
if (executionIndexes.has(observation.executionIndex)) throw new Error(`duplicate benchmark executionIndex ${observation.executionIndex} at line ${index + 1}`);
|
|
3551
3832
|
observations.push(observation);
|
|
3552
3833
|
seen.add(key);
|
|
@@ -3991,7 +4272,7 @@ function assertCompletedArtifactMatchesRun(artifact, manifest, observations, pre
|
|
|
3991
4272
|
if (canonicalJson(artifact.inputs) !== canonicalJson(expectedInputs)) throw new Error("completed benchmark result inputs do not match the run manifest");
|
|
3992
4273
|
const provenance = artifact.result.provenance;
|
|
3993
4274
|
const expectedDatasetId = config.dataset === "agentrx" ? "microsoft/AgentRx" : "NJU-LINK/CodeTraceBench";
|
|
3994
|
-
const expectedOutputAdapter = config.dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-
|
|
4275
|
+
const expectedOutputAdapter = config.dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-block";
|
|
3995
4276
|
if (provenance.id !== `${config.dataset}-real-model-analyst` || provenance.startedAt !== manifest.createdAt || !Number.isFinite(Date.parse(provenance.endedAt)) || Date.parse(provenance.endedAt) < Date.parse(provenance.startedAt) || canonicalJson(provenance.dataset) !== canonicalJson({
|
|
3996
4277
|
id: expectedDatasetId,
|
|
3997
4278
|
revision: config.datasetRevision,
|
|
@@ -4001,7 +4282,7 @@ function assertCompletedArtifactMatchesRun(artifact, manifest, observations, pre
|
|
|
4001
4282
|
if (canonicalJson(artifact.result.summaries) !== canonicalJson(expectedSummaries)) throw new Error("completed benchmark summaries do not match durable observations");
|
|
4002
4283
|
const expectedComparisons = [compareAnalystRunners(artifact.result, {
|
|
4003
4284
|
baselineRunnerId: "empty",
|
|
4004
|
-
candidateRunnerId:
|
|
4285
|
+
candidateRunnerId: config.runnerIds[1],
|
|
4005
4286
|
seed: config.seed
|
|
4006
4287
|
})];
|
|
4007
4288
|
if (canonicalJson(artifact.comparisons) !== canonicalJson(expectedComparisons)) throw new Error("completed benchmark comparisons do not match durable observations");
|
|
@@ -4134,7 +4415,7 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
|
|
|
4134
4415
|
const localIdentitySha256 = digestCanonical(localReceipt.local);
|
|
4135
4416
|
const identitySha256 = digestCanonical(identity);
|
|
4136
4417
|
const manifest = config.resume ? await readAndValidateResumeFiles(paths, identity, identitySha256, localIdentitySha256, localReceipt) : await initializeRunFiles(paths, identity, identitySha256, localIdentitySha256, localReceipt);
|
|
4137
|
-
const progress = await readProgress(paths.observations, manifest.identitySha256, prepared.selectedCaseIds, config.repetitions);
|
|
4418
|
+
const progress = await readProgress(paths.observations, manifest.identitySha256, prepared.selectedCaseIds, config.repetitions, config.analyst);
|
|
4138
4419
|
const costLedger = createRunCostLedger({
|
|
4139
4420
|
storage: fsCampaignStorage(),
|
|
4140
4421
|
runDir: paths.directory,
|
|
@@ -4147,10 +4428,10 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
|
|
|
4147
4428
|
const markdown = renderArtifactMarkdown(artifact);
|
|
4148
4429
|
await writeExclusiveOrVerify(paths.report, markdown);
|
|
4149
4430
|
printSuccessSummary(artifact, paths);
|
|
4150
|
-
return benchmarkExitCode(artifact.result);
|
|
4431
|
+
return benchmarkExitCode(artifact.result, config.analyst);
|
|
4151
4432
|
}
|
|
4152
4433
|
if (await regularFileExists(paths.report)) throw new Error(`benchmark report exists without a completed result; refusing ambiguous resume: ${paths.report}`);
|
|
4153
|
-
const createAnalystRunner = dependencies.createAnalystRunner ?? ((dataset, model) => createPublicBenchmarkRlmRunner(dataset, model));
|
|
4434
|
+
const createAnalystRunner = dependencies.createAnalystRunner ?? ((dataset, model) => config.analyst === "direct" ? createPublicBenchmarkDirectRunner(dataset, model) : createPublicBenchmarkRlmRunner(dataset, model));
|
|
4154
4435
|
const runners = [emptyPublicBenchmarkRunner(), createAnalystRunner(config.dataset, {
|
|
4155
4436
|
...config.model,
|
|
4156
4437
|
costLedger,
|
|
@@ -4172,7 +4453,7 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
|
|
|
4172
4453
|
initialObservations: progress.observations,
|
|
4173
4454
|
signal: runAbort.signal,
|
|
4174
4455
|
onObservation: async (observation) => {
|
|
4175
|
-
assertObservationAccountingComplete(observation, costLedger);
|
|
4456
|
+
assertObservationAccountingComplete(observation, costLedger, config.analyst);
|
|
4176
4457
|
await appendObservation(observation);
|
|
4177
4458
|
},
|
|
4178
4459
|
resolveEvidence: traceStoreEvidenceResolver((input) => {
|
|
@@ -4193,7 +4474,7 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
|
|
|
4193
4474
|
},
|
|
4194
4475
|
metadata: {
|
|
4195
4476
|
model: config.model.model,
|
|
4196
|
-
outputAdapter: config.dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-
|
|
4477
|
+
outputAdapter: config.dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-block",
|
|
4197
4478
|
caseSelection: prepared.selection.method,
|
|
4198
4479
|
caseSelectionSeed: config.seed,
|
|
4199
4480
|
selectionStratified: prepared.selection.stratified,
|
|
@@ -4211,11 +4492,11 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
|
|
|
4211
4492
|
}
|
|
4212
4493
|
assertCostLedgerFinalizable(costLedger);
|
|
4213
4494
|
result.provenance.startedAt = manifest.createdAt;
|
|
4214
|
-
const persisted = await readProgress(paths.observations, manifest.identitySha256, prepared.selectedCaseIds, config.repetitions);
|
|
4495
|
+
const persisted = await readProgress(paths.observations, manifest.identitySha256, prepared.selectedCaseIds, config.repetitions, config.analyst);
|
|
4215
4496
|
assertSameObservations(result.observations, persisted.observations);
|
|
4216
4497
|
const comparisons = [compareAnalystRunners(result, {
|
|
4217
4498
|
baselineRunnerId: "empty",
|
|
4218
|
-
candidateRunnerId:
|
|
4499
|
+
candidateRunnerId: config.analyst,
|
|
4219
4500
|
seed: config.seed
|
|
4220
4501
|
})];
|
|
4221
4502
|
const codeTraceCalibration = config.dataset === "codetracebench" ? summarizeCodeTraceCalibration(result) : void 0;
|
|
@@ -4260,7 +4541,7 @@ async function executeAnalystBenchmarkCommand(config, dependencies) {
|
|
|
4260
4541
|
await writeExclusiveOrVerify(paths.result, `${JSON.stringify(artifact, null, 2)}\n`);
|
|
4261
4542
|
await writeExclusiveOrVerify(paths.report, markdown);
|
|
4262
4543
|
printSuccessSummary(artifact, paths);
|
|
4263
|
-
return benchmarkExitCode(result);
|
|
4544
|
+
return benchmarkExitCode(result, config.analyst);
|
|
4264
4545
|
}
|
|
4265
4546
|
const NON_SCORABLE_COST_ERRORS = /* @__PURE__ */ new Set([
|
|
4266
4547
|
"CostAccountingIncompleteError",
|
|
@@ -4270,27 +4551,33 @@ const NON_SCORABLE_COST_ERRORS = /* @__PURE__ */ new Set([
|
|
|
4270
4551
|
"CostReceiptCaptureError",
|
|
4271
4552
|
"CostReservationExceededError"
|
|
4272
4553
|
]);
|
|
4273
|
-
function assertObservationAccountingComplete(observation, costLedger) {
|
|
4554
|
+
function assertObservationAccountingComplete(observation, costLedger, analystRunnerId) {
|
|
4274
4555
|
if (observation.error && NON_SCORABLE_COST_ERRORS.has(observation.error.class)) throw new CostAccountingIncompleteError(`Analyst benchmark stopped before scoring: ${observation.error.message}`);
|
|
4275
|
-
if (observation.runnerId !==
|
|
4276
|
-
const
|
|
4277
|
-
channel: "analyst",
|
|
4278
|
-
tags: {
|
|
4279
|
-
benchmarkCaseId: observation.caseId,
|
|
4280
|
-
benchmarkRepetition: String(observation.repetition)
|
|
4281
|
-
}
|
|
4282
|
-
});
|
|
4283
|
-
if (!summary.accountingComplete || summary.pendingCalls > 0 || summary.unresolvedCalls > 0) throw accountingError(costLedger, "the recursive analyst has incomplete cost accounting", {
|
|
4556
|
+
if (observation.runnerId !== analystRunnerId) return;
|
|
4557
|
+
const filter = {
|
|
4284
4558
|
channel: "analyst",
|
|
4285
4559
|
tags: {
|
|
4286
4560
|
benchmarkCaseId: observation.caseId,
|
|
4287
4561
|
benchmarkRepetition: String(observation.repetition)
|
|
4288
4562
|
}
|
|
4289
|
-
}
|
|
4563
|
+
};
|
|
4564
|
+
if (!costAccountingIsTrustworthy(costLedger.summary(filter))) throw accountingError(costLedger, "the recursive analyst has incomplete cost accounting", filter);
|
|
4565
|
+
}
|
|
4566
|
+
const BUDGET_BREACH_REASON = /exceeding its enforced maximum/;
|
|
4567
|
+
/**
|
|
4568
|
+
* Cost accounting is trustworthy when every call resolved and none breached its
|
|
4569
|
+
* budget. A recursive analyst on a real provider will occasionally receive a
|
|
4570
|
+
* settled response whose usage the provider omitted; that call is honestly
|
|
4571
|
+
* recorded as unknown and excluded from the reported cost, so it does not
|
|
4572
|
+
* invalidate a completed run. A call left pending, one lost, or one charged
|
|
4573
|
+
* beyond its maximum is a genuine integrity failure and still halts.
|
|
4574
|
+
*/
|
|
4575
|
+
function costAccountingIsTrustworthy(summary) {
|
|
4576
|
+
if (summary.pendingCalls > 0 || summary.unresolvedCalls > 0) return false;
|
|
4577
|
+
return !summary.incompleteReasons.some((reason) => BUDGET_BREACH_REASON.test(reason));
|
|
4290
4578
|
}
|
|
4291
4579
|
function assertCostLedgerFinalizable(costLedger) {
|
|
4292
|
-
|
|
4293
|
-
if (!summary.accountingComplete || summary.pendingCalls > 0 || summary.unresolvedCalls > 0) throw accountingError(costLedger, "the run has pending or incomplete cost entries");
|
|
4580
|
+
if (!costAccountingIsTrustworthy(costLedger.summary())) throw accountingError(costLedger, "the run has pending or budget-breaching cost entries");
|
|
4294
4581
|
}
|
|
4295
4582
|
function accountingError(costLedger, reason, filter) {
|
|
4296
4583
|
const details = costLedger.summary(filter).incompleteReasons.slice(0, 3).join("; ");
|
|
@@ -4302,6 +4589,9 @@ Run the recursive DSPy trace analyst against public AgentRx or CodeTraceBench la
|
|
|
4302
4589
|
|
|
4303
4590
|
Required:
|
|
4304
4591
|
--dataset agentrx|codetracebench
|
|
4592
|
+
--analyst dspy-rlm|direct Scored analyst. Default: dspy-rlm.
|
|
4593
|
+
'direct' is the retired one-shot runner that
|
|
4594
|
+
produced the published evidence.
|
|
4305
4595
|
--labels <dataset.json|dataset.jsonl>
|
|
4306
4596
|
--trace-dir <one-trace-per-file OTLP JSONL directory>
|
|
4307
4597
|
--artifact-dir <extracted artifact root> Required for CodeTraceBench
|
|
@@ -4318,7 +4608,7 @@ Controls:
|
|
|
4318
4608
|
--seed <integer> Case-selection and comparison seed. Default: 0
|
|
4319
4609
|
--concurrency <positive integer> Parallel benchmark jobs. Default: 1
|
|
4320
4610
|
--repetitions <positive integer> Runs per case and runner. Default: 1
|
|
4321
|
-
--max-output-tokens <positive> Model output limit per call. Default:
|
|
4611
|
+
--max-output-tokens <positive> Model output limit per call. Default: 16384
|
|
4322
4612
|
--python <executable> Python with agent-eval-rpc[dspy]. Default: python
|
|
4323
4613
|
--timeout-ms <positive> Model analyst deadline per case. Default: 300000
|
|
4324
4614
|
--max-cost-usd <positive> Run-wide spend limit. Default: 5
|
|
@@ -4344,8 +4634,11 @@ function parseCommandConfig(argv, env) {
|
|
|
4344
4634
|
if (!apiKey) throw new Error(`--api-key-env points to an empty or missing variable: ${apiKeyEnv}`);
|
|
4345
4635
|
const maxCostUsd = positiveFiniteFlag(flags, "max-cost-usd", 5);
|
|
4346
4636
|
const python = flags.get("python")?.trim();
|
|
4637
|
+
const analyst = flags.get("analyst")?.trim() ?? "dspy-rlm";
|
|
4638
|
+
if (analyst !== "dspy-rlm" && analyst !== "direct") throw new Error("--analyst must be 'dspy-rlm' or 'direct'");
|
|
4347
4639
|
return {
|
|
4348
4640
|
dataset,
|
|
4641
|
+
analyst,
|
|
4349
4642
|
labelsPath: requiredFlag(flags, "labels"),
|
|
4350
4643
|
traceDir: requiredFlag(flags, "trace-dir"),
|
|
4351
4644
|
...artifactDir ? { artifactDir } : {},
|
|
@@ -4356,7 +4649,7 @@ function parseCommandConfig(argv, env) {
|
|
|
4356
4649
|
baseUrl: openAiCompatibleBaseUrl(requiredFlag(flags, "base-url")),
|
|
4357
4650
|
apiKey,
|
|
4358
4651
|
model: requiredFlag(flags, "model"),
|
|
4359
|
-
maxOutputTokens: positiveFlag(flags, "max-output-tokens",
|
|
4652
|
+
maxOutputTokens: positiveFlag(flags, "max-output-tokens", 16384),
|
|
4360
4653
|
timeoutMs: positiveFlag(flags, "timeout-ms", 3e5),
|
|
4361
4654
|
maxCostUsdPerAnalysis: maxCostUsd,
|
|
4362
4655
|
...python ? { dspyRlm: { runner: { command: python } } } : {}
|
|
@@ -4396,6 +4689,7 @@ function parseFlags(argv) {
|
|
|
4396
4689
|
const KNOWN_FLAGS = /* @__PURE__ */ new Set([
|
|
4397
4690
|
"resume",
|
|
4398
4691
|
"dataset",
|
|
4692
|
+
"analyst",
|
|
4399
4693
|
"labels",
|
|
4400
4694
|
"trace-dir",
|
|
4401
4695
|
"artifact-dir",
|
|
@@ -4519,8 +4813,8 @@ function renderArtifactMarkdown(artifact) {
|
|
|
4519
4813
|
const verificationMarkdown = artifact.inputs.dataset === "codetracebench" ? `\n\n${renderVerificationAvailability(artifact.inputs.verificationAvailability)}` : "";
|
|
4520
4814
|
return `${renderAnalystBenchmarkMarkdown(artifact.result, artifact.comparisons).trimEnd()}${calibrationMarkdown}${verificationMarkdown}\n\n${renderSelectionMarkdown(artifact.inputs.selection.report)}\n`;
|
|
4521
4815
|
}
|
|
4522
|
-
function benchmarkExitCode(result) {
|
|
4523
|
-
return result.summaries.find((summary) => summary.runnerId ===
|
|
4816
|
+
function benchmarkExitCode(result, analystRunnerId) {
|
|
4817
|
+
return result.summaries.find((summary) => summary.runnerId === analystRunnerId)?.failedRuns ? 2 : 0;
|
|
4524
4818
|
}
|
|
4525
4819
|
function printSuccessSummary(artifact, paths) {
|
|
4526
4820
|
const failures = artifact.result.summaries.reduce((total, summary) => total + summary.failedRuns, 0);
|
|
@@ -4532,6 +4826,6 @@ function shellQuote(value) {
|
|
|
4532
4826
|
return /^[a-zA-Z0-9_./:@%+=,-]+$/.test(value) ? value : `'${value.replaceAll("'", "'\\''")}'`;
|
|
4533
4827
|
}
|
|
4534
4828
|
//#endregion
|
|
4535
|
-
export {
|
|
4829
|
+
export { ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as A, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as B, publicBenchmarkSystemPrompt as C, parseVerificationOutcome as D, loadCodeTraceVerificationArtifacts as E, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as F, summarizeAgentRxCalibration as G, ANALYST_BENCHMARK_OBSERVATIONS_FILE as H, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as I, agentRxBenchmarkCase as J, codeTraceBenchCase as K, analystBenchmarkDependencyLockDigest as L, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as M, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as N, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as O, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as P, normalizeBenchmarkLabel as Q, analystBenchmarkImplementationDigest as R, publicBenchmarkRlmInstructions as S, appendVerificationArtifactsToOtlp as T, AGENT_RX_UPSTREAM_REVISION as U, ANALYST_BENCHMARK_MANIFEST_FILE as V, renderAgentRxCalibrationMarkdown as W, normalizeAgentRxCategory as X, agentRxPredictionsToFindings as Y, roundAgentRxStep as Z, expandCodeTraceFailureBlocks as _, renderCodeTraceCalibrationMarkdown as a, MAX_INCORRECT_BLOCK_STEPS as b, createPublicBenchmarkRlmRunner as c, preparePublicAnalystBenchmark as d, publicBenchmarkDistributions as f, emptyPublicBenchmarkRunner as g, adaptPublicBenchmarkFindings as h, readAnalystBenchmarkArtifact as i, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as j, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as k, createPublicBenchmarkDirectRunner as l, selectPublicBenchmarkRows as m, runAnalystBenchmarkCommand as n, summarizeCodeTraceCalibration as o, publicBenchmarkSelectionReport as p, codeTracerPredictionsToFindings as q, renderAnalystBenchmarkMarkdown as r, compareAnalystRunners as s, ANALYST_BENCHMARK_HELP as t, loadPublicBenchmarkRows as u, CODE_TRACE_BENCH_ANALYST_PROMPT as v, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as w, publicBenchmarkProtocolSha256 as x, MAX_INCORRECT_BLOCKS as y, ANALYST_BENCHMARK_COST_LEDGER_FILE as z };
|
|
4536
4830
|
|
|
4537
|
-
//# sourceMappingURL=benchmark-command-
|
|
4831
|
+
//# sourceMappingURL=benchmark-command-CK0UnXAD.js.map
|