@tangle-network/agent-eval 0.137.0 → 0.138.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +48 -0
- package/README.md +33 -0
- package/dist/analyst/index.d.ts +473 -39
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +11 -593
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-PVtnfjvA.d.ts → analyze-runs-CPYxfPWT.d.ts} +5 -5
- package/dist/{analyze-runs-PVtnfjvA.d.ts.map → analyze-runs-CPYxfPWT.d.ts.map} +1 -1
- package/dist/{benchmark-YDrpumqB.js → benchmark-D8dkki-J.js} +299 -159
- package/dist/benchmark-D8dkki-J.js.map +1 -0
- package/dist/{benchmark-CHX4orG7.d.ts → benchmark-DlQgU_XI.d.ts} +67 -15
- package/dist/benchmark-DlQgU_XI.d.ts.map +1 -0
- package/dist/benchmark-command-CMqVqReF.js +4332 -0
- package/dist/benchmark-command-CMqVqReF.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-DCLkQOmc.js → benchmarks-BJ_xK5rQ.js} +4 -3
- package/dist/{benchmarks-DCLkQOmc.js.map → benchmarks-BJ_xK5rQ.js.map} +1 -1
- package/dist/campaign/index.d.ts +5 -5
- package/dist/campaign/index.js +3 -3
- package/dist/{campaign-lgObcHFC.js → campaign-BIBS-NHV.js} +16 -9
- package/dist/campaign-BIBS-NHV.js.map +1 -0
- package/dist/cli.js +9 -2
- package/dist/cli.js.map +1 -1
- package/dist/{client-C8L6h6Wf.d.ts → client-BwPKohkJ.d.ts} +4 -4
- package/dist/{client-C8L6h6Wf.d.ts.map → client-BwPKohkJ.d.ts.map} +1 -1
- package/dist/{completion-verifier-DSyRNVzU.d.ts → completion-verifier-B4-IMYcS.d.ts} +3 -3
- package/dist/{completion-verifier-DSyRNVzU.d.ts.map → completion-verifier-B4-IMYcS.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +10 -10
- package/dist/contract/index.js +8 -8
- package/dist/control.d.ts +2 -2
- package/dist/{cost-ledger-D2o6JOrL.d.ts → cost-ledger-B1D3COAc.d.ts} +5 -4
- package/dist/{cost-ledger-D2o6JOrL.d.ts.map → cost-ledger-B1D3COAc.d.ts.map} +1 -1
- package/dist/{cost-ledger-D-5_-dhi.js → cost-ledger-CHDLA0Ss.js} +90 -45
- package/dist/cost-ledger-CHDLA0Ss.js.map +1 -0
- package/dist/{default-registry-Dc5D_Loc.d.ts → default-registry-PUhIVRWz.d.ts} +18 -5
- package/dist/default-registry-PUhIVRWz.d.ts.map +1 -0
- package/dist/{default-registry-CLXbRt0f.js → default-registry-lp5R0lve.js} +1503 -258
- package/dist/default-registry-lp5R0lve.js.map +1 -0
- package/dist/{eval-campaign-CHqfLnff.js → eval-campaign-9MozgKL7.js} +2 -2
- package/dist/{eval-campaign-CHqfLnff.js.map → eval-campaign-9MozgKL7.js.map} +1 -1
- package/dist/exact-types-Dpw2LeHA.d.ts +234 -0
- package/dist/exact-types-Dpw2LeHA.d.ts.map +1 -0
- package/dist/{extract-usage-p-56bh8q.js → extract-usage-CS391dOE.js} +2 -2
- package/dist/{extract-usage-p-56bh8q.js.map → extract-usage-CS391dOE.js.map} +1 -1
- package/dist/{feedback-trajectory-N_F0PwHz.d.ts → feedback-trajectory-CoNep7rl.d.ts} +3 -2
- package/dist/feedback-trajectory-CoNep7rl.d.ts.map +1 -0
- package/dist/fuzz.d.ts +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-U3RHOShi.d.ts → index-B2-IxCMB.d.ts} +2 -2
- package/dist/{index-U3RHOShi.d.ts.map → index-B2-IxCMB.d.ts.map} +1 -1
- package/dist/{index-BnP1QJUv.d.ts → index-CjVYlVBK.d.ts} +5 -5
- package/dist/{index-BnP1QJUv.d.ts.map → index-CjVYlVBK.d.ts.map} +1 -1
- package/dist/{index-C-Pr4OWg.d.ts → index-D0cxAdaV.d.ts} +11 -10
- package/dist/index-D0cxAdaV.d.ts.map +1 -0
- package/dist/index-DEb46kc6.d.ts.map +1 -1
- package/dist/{index-DRNl6g_N.d.ts → index-sMN_hI4E.d.ts} +3 -3
- package/dist/{index-DRNl6g_N.d.ts.map → index-sMN_hI4E.d.ts.map} +1 -1
- package/dist/index.d.ts +24 -23
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +19 -353
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-B9ooYH_g.d.ts → insight-report-CXd8VBDR.d.ts} +4 -4
- package/dist/{insight-report-B9ooYH_g.d.ts.map → insight-report-CXd8VBDR.d.ts.map} +1 -1
- package/dist/{integrity-CKxosZ5Z.d.ts → integrity-B-MLFz0I.d.ts} +2 -2
- package/dist/{integrity-CKxosZ5Z.d.ts.map → integrity-B-MLFz0I.d.ts.map} +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{ledger-core-t6sItivm.js → ledger-core-C0Yx1I14.js} +220 -27
- package/dist/ledger-core-C0Yx1I14.js.map +1 -0
- package/dist/{llm-client-DKB25jV8.js → llm-client-Cj3c7PEm.js} +5 -5
- package/dist/llm-client-Cj3c7PEm.js.map +1 -0
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{proposal-findings-DCawte-y.js → proposal-findings-2GIUo1et.js} +2 -68
- package/dist/proposal-findings-2GIUo1et.js.map +1 -0
- package/dist/{registry-BdM7SuTr.d.ts → registry-C4yJTza7.d.ts} +60 -6
- package/dist/registry-C4yJTza7.d.ts.map +1 -0
- package/dist/{release-report-CofgVNZt.d.ts → release-report-CoyvyLBs.d.ts} +3 -3
- package/dist/{release-report-CofgVNZt.d.ts.map → release-report-CoyvyLBs.d.ts.map} +1 -1
- package/dist/{replay-Bju0T8Ls.js → replay-Cb-4Vf0k.js} +8 -7
- package/dist/replay-Cb-4Vf0k.js.map +1 -0
- package/dist/{replay-K8FaC0CB.d.ts → replay-DbIYwso6.d.ts} +7 -7
- package/dist/{replay-K8FaC0CB.d.ts.map → replay-DbIYwso6.d.ts.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-Da0Wj-bt.d.ts → researcher-BCeOEjtR.d.ts} +5 -5
- package/dist/{researcher-Da0Wj-bt.d.ts.map → researcher-BCeOEjtR.d.ts.map} +1 -1
- package/dist/{reward-hacking-CQ3hTCO3.d.ts → reward-hacking-sE2l_NV6.d.ts} +2 -2
- package/dist/{reward-hacking-CQ3hTCO3.d.ts.map → reward-hacking-sE2l_NV6.d.ts.map} +1 -1
- package/dist/rl.d.ts +5 -5
- package/dist/rl.js +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/{rubric-predictive-validity-C4sztLR3.d.ts → rubric-predictive-validity-w2klGv1u.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C4sztLR3.d.ts.map → rubric-predictive-validity-w2klGv1u.d.ts.map} +1 -1
- package/dist/{run-evidence-BDIircdA.d.ts → run-evidence-CbE0A8Xg.d.ts} +3 -3
- package/dist/{run-evidence-BDIircdA.d.ts.map → run-evidence-CbE0A8Xg.d.ts.map} +1 -1
- package/dist/{run-record-BPCa2rQ8.d.ts → run-record-DwHMk1Ai.d.ts} +2 -2
- package/dist/{run-record-BPCa2rQ8.d.ts.map → run-record-DwHMk1Ai.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-Bz64IckK.js → semantic-concept-judge-DYXDPZW0.js} +11 -5
- package/dist/semantic-concept-judge-DYXDPZW0.js.map +1 -0
- package/dist/{server-KjXZZUDX.js → server-DLEvyW2z.js} +3 -3
- package/dist/{server-KjXZZUDX.js.map → server-DLEvyW2z.js.map} +1 -1
- package/dist/single-run-lock-D_bS5xhj.js +318 -0
- package/dist/single-run-lock-D_bS5xhj.js.map +1 -0
- package/dist/{skill-usage-CFDLLlhF.d.ts → skill-usage-Bv3G4VkA.d.ts} +18 -8
- package/dist/skill-usage-Bv3G4VkA.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-f4o9sUT4.js → skillopt-optimization-method-CjKMZy0d.js} +7 -182
- package/dist/skillopt-optimization-method-CjKMZy0d.js.map +1 -0
- package/dist/{skillopt-optimization-method-BpbnlvAZ.d.ts → skillopt-optimization-method-CzfnA8O-.d.ts} +10 -10
- package/dist/{skillopt-optimization-method-BpbnlvAZ.d.ts.map → skillopt-optimization-method-CzfnA8O-.d.ts.map} +1 -1
- package/dist/{statistics-_7P642CN.d.ts → statistics-mf70aXKp.d.ts} +2 -2
- package/dist/{statistics-_7P642CN.d.ts.map → statistics-mf70aXKp.d.ts.map} +1 -1
- package/dist/{tools-DZk2Jn64.js → store-otlp-BenKynPE.js} +4 -192
- package/dist/store-otlp-BenKynPE.js.map +1 -0
- package/dist/{summary-report-DHipz9Kx.d.ts → summary-report-BKinV4yD.d.ts} +3 -3
- package/dist/{summary-report-DHipz9Kx.d.ts.map → summary-report-BKinV4yD.d.ts.map} +1 -1
- package/dist/tools-DZGdROtG.js +255 -0
- package/dist/tools-DZGdROtG.js.map +1 -0
- package/dist/traces.d.ts +5 -5
- package/dist/traces.js +4 -3
- package/dist/{types-CTvKfr5F.d.ts → types-5q2T25iW.d.ts} +2 -2
- package/dist/{types-CTvKfr5F.d.ts.map → types-5q2T25iW.d.ts.map} +1 -1
- package/dist/{types-CKswbJGO.d.ts → types-BtJhn8v6.d.ts} +4 -4
- package/dist/{types-CKswbJGO.d.ts.map → types-BtJhn8v6.d.ts.map} +1 -1
- package/dist/{types-CTGbIm57.d.ts → types-zFYez3PK.d.ts} +5 -5
- package/dist/{types-CTGbIm57.d.ts.map → types-zFYez3PK.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/trace-analysis.md +123 -3
- package/package.json +5 -3
- package/dist/benchmark-CHX4orG7.d.ts.map +0 -1
- package/dist/benchmark-YDrpumqB.js.map +0 -1
- package/dist/campaign-lgObcHFC.js.map +0 -1
- package/dist/concurrency-MUjT7VjM.js +0 -109
- package/dist/concurrency-MUjT7VjM.js.map +0 -1
- package/dist/cost-ledger-D-5_-dhi.js.map +0 -1
- package/dist/default-registry-CLXbRt0f.js.map +0 -1
- package/dist/default-registry-Dc5D_Loc.d.ts.map +0 -1
- package/dist/feedback-trajectory-N_F0PwHz.d.ts.map +0 -1
- package/dist/index-C-Pr4OWg.d.ts.map +0 -1
- package/dist/ledger-core-t6sItivm.js.map +0 -1
- package/dist/llm-client-DKB25jV8.js.map +0 -1
- package/dist/proposal-findings-DCawte-y.js.map +0 -1
- package/dist/registry-BdM7SuTr.d.ts.map +0 -1
- package/dist/replay-Bju0T8Ls.js.map +0 -1
- package/dist/semantic-concept-judge-Bz64IckK.js.map +0 -1
- package/dist/skill-usage-CFDLLlhF.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-f4o9sUT4.js.map +0 -1
- package/dist/tools-DZk2Jn64.js.map +0 -1
|
@@ -0,0 +1,4332 @@
|
|
|
1
|
+
import { c as ValidationError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
|
|
2
|
+
import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-CHDLA0Ss.js";
|
|
3
|
+
import { c as callLlmJson, f as maximumChargeForLlmRequest, l as costReceiptFromLlm, r as LlmResponseError, t as LlmCallError, u as costReceiptFromLlmError } from "./llm-client-Cj3c7PEm.js";
|
|
4
|
+
import { d as TRACE_ANALYSIS_LIMITS, i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-BenKynPE.js";
|
|
5
|
+
import { d as usageReceiptFromCostLedger, m as makeFinding, n as createRunCostLedger, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-D_bS5xhj.js";
|
|
6
|
+
import { _ as canonicalString, c as writeLedgerFileAtomically, s as withLedgerFileLock, v as hashCanonical } from "./ledger-core-C0Yx1I14.js";
|
|
7
|
+
import { E as pairedBootstrap } from "./statistics-ByxzSiOM.js";
|
|
8
|
+
import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-D8dkki-J.js";
|
|
9
|
+
import { z } from "zod";
|
|
10
|
+
import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
|
|
11
|
+
import * as nodePath from "node:path";
|
|
12
|
+
import { dirname, isAbsolute, relative, resolve, sep } from "node:path";
|
|
13
|
+
import { createHash, randomUUID } from "node:crypto";
|
|
14
|
+
import { link, lstat, mkdir, open, readFile, readdir, realpath, stat, unlink } from "node:fs/promises";
|
|
15
|
+
import { arch, platform } from "node:os";
|
|
16
|
+
import { TextDecoder as TextDecoder$1 } from "node:util";
|
|
17
|
+
//#region src/analyst/benchmark-dataset-utils.ts
|
|
18
|
+
function normalizeBenchmarkLabel(value) {
|
|
19
|
+
const normalized = nonEmpty$1(value, "benchmark label").toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g, "");
|
|
20
|
+
if (!normalized) throw new TypeError("benchmark label must contain letters or digits");
|
|
21
|
+
return normalized;
|
|
22
|
+
}
|
|
23
|
+
function predictionConfidence(value) {
|
|
24
|
+
const confidence = value ?? .5;
|
|
25
|
+
if (!Number.isFinite(confidence) || confidence < 0 || confidence > 1) throw new RangeError("upstream prediction confidence must be between 0 and 1");
|
|
26
|
+
return confidence;
|
|
27
|
+
}
|
|
28
|
+
function assertStepWithinRange(step, stepCount, field) {
|
|
29
|
+
if (stepCount === void 0) return;
|
|
30
|
+
const count = positiveStep(stepCount, `${field} stepCount`);
|
|
31
|
+
if (step > count) throw new RangeError(`${field} step ${step} exceeds stepCount ${count}`);
|
|
32
|
+
}
|
|
33
|
+
function defaultStepUri(trajectoryId, step) {
|
|
34
|
+
return `trace://${encodeURIComponent(trajectoryId)}/span/step-${step}`;
|
|
35
|
+
}
|
|
36
|
+
function isRecord$2(value) {
|
|
37
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
38
|
+
}
|
|
39
|
+
function positiveStep(value, field) {
|
|
40
|
+
if (!Number.isSafeInteger(value) || value < 1) throw new RangeError(`${field} must be a positive safe integer`);
|
|
41
|
+
return value;
|
|
42
|
+
}
|
|
43
|
+
function externalId(value, field) {
|
|
44
|
+
if (typeof value !== "string" && typeof value !== "number") throw new TypeError(`${field} must be a string or number`);
|
|
45
|
+
if (typeof value === "number" && !Number.isSafeInteger(value)) throw new TypeError(`${field} must be a safe integer when numeric`);
|
|
46
|
+
return nonEmpty$1(String(value), field);
|
|
47
|
+
}
|
|
48
|
+
function nonEmpty$1(value, field) {
|
|
49
|
+
if (!value.trim()) throw new TypeError(`${field} must not be empty`);
|
|
50
|
+
return value;
|
|
51
|
+
}
|
|
52
|
+
//#endregion
|
|
53
|
+
//#region src/analyst/benchmark-dataset-agentrx.ts
|
|
54
|
+
function agentRxBenchmarkCase(row, input, options = {}) {
|
|
55
|
+
const trajectoryId = externalId(row.trajectory_id, "AgentRx trajectory_id");
|
|
56
|
+
if (!Array.isArray(row.failures) || row.failures.length === 0) throw new TypeError(`AgentRx trajectory '${trajectoryId}' must contain failures`);
|
|
57
|
+
if (row.num_failures !== void 0 && row.num_failures !== row.failures.length) throw new TypeError(`AgentRx trajectory '${trajectoryId}' declares ${row.num_failures} failures but contains ${row.failures.length}`);
|
|
58
|
+
const rootCauseId = externalId(row.root_cause_failure_id ?? row.root_cause?.failure_id, `AgentRx trajectory '${trajectoryId}' root cause failure id`);
|
|
59
|
+
const failureIds = /* @__PURE__ */ new Set();
|
|
60
|
+
const failureMetadata = [];
|
|
61
|
+
const evidenceKind = options.evidenceKind ?? "span";
|
|
62
|
+
const uri = options.stepUri ?? defaultStepUri;
|
|
63
|
+
const allIssues = row.failures.map((failure) => {
|
|
64
|
+
const failureId = externalId(failure.failure_id, `AgentRx trajectory '${trajectoryId}' failure id`);
|
|
65
|
+
if (failureIds.has(failureId)) throw new TypeError(`AgentRx trajectory '${trajectoryId}' repeats failure id '${failureId}'`);
|
|
66
|
+
failureIds.add(failureId);
|
|
67
|
+
const step = positiveStep(failure.step_number, `AgentRx trajectory '${trajectoryId}'`);
|
|
68
|
+
const evidence = [{
|
|
69
|
+
kind: evidenceKind,
|
|
70
|
+
uri: uri(trajectoryId, step)
|
|
71
|
+
}];
|
|
72
|
+
if (typeof failure.failure_category !== "string") throw new TypeError(`AgentRx trajectory '${trajectoryId}' failure '${failureId}' category must be a string`);
|
|
73
|
+
const category = normalizeAgentRxCategory(failure.failure_category);
|
|
74
|
+
if (!AGENT_RX_TAXONOMY_BY_LABEL.has(category)) throw new RangeError(`AgentRx trajectory '${trajectoryId}' failure '${failureId}' category '${failure.failure_category}' is outside the AgentRx taxonomy`);
|
|
75
|
+
failureMetadata.push({
|
|
76
|
+
id: failureId,
|
|
77
|
+
step,
|
|
78
|
+
category
|
|
79
|
+
});
|
|
80
|
+
return {
|
|
81
|
+
id: failureId,
|
|
82
|
+
areas: [category],
|
|
83
|
+
...failureId === rootCauseId && (options.target ?? "root-cause") === "root-cause" ? {} : { evidence },
|
|
84
|
+
criticalEvidence: failureId === rootCauseId ? evidence : void 0
|
|
85
|
+
};
|
|
86
|
+
});
|
|
87
|
+
if (!failureIds.has(rootCauseId)) throw new TypeError(`AgentRx trajectory '${trajectoryId}' root cause '${rootCauseId}' is not in failures`);
|
|
88
|
+
if (options.stepCount !== void 0 && row.failures.some((failure) => failure.step_number > options.stepCount)) throw new RangeError(`AgentRx trajectory '${trajectoryId}' contains a failure beyond stepCount ${options.stepCount}`);
|
|
89
|
+
const expectedIssues = (options.target ?? "root-cause") === "root-cause" ? allIssues.filter((issue) => issue.id === rootCauseId) : allIssues;
|
|
90
|
+
const rootCause = failureMetadata.find((failure) => failure.id === rootCauseId);
|
|
91
|
+
const orderedFailures = [...failureMetadata].sort((left, right) => left.step - right.step || left.id.localeCompare(right.id));
|
|
92
|
+
return {
|
|
93
|
+
id: `agentrx:${trajectoryId}`,
|
|
94
|
+
clusterId: `agentrx:${trajectoryId}`,
|
|
95
|
+
labelState: "positive",
|
|
96
|
+
input,
|
|
97
|
+
expectedIssues,
|
|
98
|
+
labeledEvidence: expectedIssues.flatMap((issue) => issue.evidence ?? issue.criticalEvidence ?? []),
|
|
99
|
+
tags: ["agentrx"],
|
|
100
|
+
metadata: {
|
|
101
|
+
benchmark: "AgentRx",
|
|
102
|
+
trajectoryId,
|
|
103
|
+
failureSummary: row.failure_summary,
|
|
104
|
+
rootCauseReason: row.root_cause_reason ?? row.root_cause?.reason_for_root_cause,
|
|
105
|
+
annotatedFailures: row.failures.length,
|
|
106
|
+
target: options.target ?? "root-cause",
|
|
107
|
+
rootCauseStep: rootCause.step,
|
|
108
|
+
rootCauseCategory: rootCause.category,
|
|
109
|
+
allFailureCategories: [...new Set(failureMetadata.map((failure) => failure.category))].sort(),
|
|
110
|
+
earliestFailureCategory: orderedFailures[0].category,
|
|
111
|
+
terminalFailureCategory: orderedFailures.at(-1).category,
|
|
112
|
+
...options.stepCount === void 0 ? {} : { trajectoryLength: options.stepCount }
|
|
113
|
+
}
|
|
114
|
+
};
|
|
115
|
+
}
|
|
116
|
+
/** Translate AgentRx `Report.to_dict()` output or its `failures` array into findings. */
|
|
117
|
+
function agentRxPredictionsToFindings(trajectoryIdValue, output, options = {}) {
|
|
118
|
+
const trajectoryId = externalId(trajectoryIdValue, "AgentRx prediction trajectory id");
|
|
119
|
+
const parsed = parseAgentRxPredictions(output, trajectoryId);
|
|
120
|
+
for (const prediction of parsed.predictions) {
|
|
121
|
+
assertStepWithinRange(prediction.step_number, parsed.report?.trajectory_length, `AgentRx prediction '${trajectoryId}' report`);
|
|
122
|
+
assertStepWithinRange(prediction.step_number, options.stepCount, `AgentRx prediction '${trajectoryId}'`);
|
|
123
|
+
}
|
|
124
|
+
const consensus = agentRxConsensus(parsed, trajectoryId);
|
|
125
|
+
if (consensus.failureCase === 0) return [];
|
|
126
|
+
const confidence = predictionConfidence(options.confidence);
|
|
127
|
+
const uri = options.stepUri ?? defaultStepUri;
|
|
128
|
+
assertStepWithinRange(consensus.step, options.stepCount, `AgentRx prediction '${trajectoryId}'`);
|
|
129
|
+
const area = AGENT_RX_TAXONOMY.get(consensus.failureCase);
|
|
130
|
+
return [makeFinding({
|
|
131
|
+
analyst_id: options.analystId ?? "agentrx",
|
|
132
|
+
produced_at: options.producedAt,
|
|
133
|
+
area,
|
|
134
|
+
subject: "root-cause",
|
|
135
|
+
claim: `AgentRx classified step ${consensus.step} as ${area}.`,
|
|
136
|
+
id_basis: `${area}:${consensus.step}`,
|
|
137
|
+
rationale: consensus.representative.description,
|
|
138
|
+
severity: "high",
|
|
139
|
+
confidence,
|
|
140
|
+
evidence_refs: [{
|
|
141
|
+
kind: options.evidenceKind ?? "span",
|
|
142
|
+
uri: uri(trajectoryId, consensus.step)
|
|
143
|
+
}],
|
|
144
|
+
metadata: {
|
|
145
|
+
upstream: "AgentRx",
|
|
146
|
+
failure_case: consensus.failureCase,
|
|
147
|
+
step: consensus.step,
|
|
148
|
+
step_mean: consensus.stepMean,
|
|
149
|
+
judge_votes: parsed.predictions.length,
|
|
150
|
+
consensus_votes: consensus.votes,
|
|
151
|
+
category_agreement: consensus.votes / parsed.predictions.length,
|
|
152
|
+
...consensus.representative.checklist_reasoning === void 0 || consensus.representative.checklist_reasoning === null ? {} : { checklist_reasoning: consensus.representative.checklist_reasoning }
|
|
153
|
+
}
|
|
154
|
+
})];
|
|
155
|
+
}
|
|
156
|
+
const AGENT_RX_TAXONOMY = /* @__PURE__ */ new Map([
|
|
157
|
+
[1, "instruction-plan-adherence-failure"],
|
|
158
|
+
[2, "invention-of-new-information"],
|
|
159
|
+
[3, "invalid-invocation"],
|
|
160
|
+
[4, "misinterpretation-of-tool-output-handoff-failure"],
|
|
161
|
+
[5, "intent-plan-misalignment"],
|
|
162
|
+
[6, "underspecified-user-intent"],
|
|
163
|
+
[7, "intent-not-supported"],
|
|
164
|
+
[8, "guardrails-triggered"],
|
|
165
|
+
[9, "system-failure"],
|
|
166
|
+
[10, "inconclusive"]
|
|
167
|
+
]);
|
|
168
|
+
const AGENT_RX_CATEGORY_ALIASES = /* @__PURE__ */ new Map([["instruction-adherence-failure", "instruction-plan-adherence-failure"], ["misinterpretation-of-tool-output", "misinterpretation-of-tool-output-handoff-failure"]]);
|
|
169
|
+
const AGENT_RX_TAXONOMY_BY_LABEL = new Map([...AGENT_RX_TAXONOMY].map(([failureCase, label]) => [label, failureCase]));
|
|
170
|
+
function normalizeAgentRxCategory(value) {
|
|
171
|
+
const normalized = normalizeBenchmarkLabel(value);
|
|
172
|
+
return AGENT_RX_CATEGORY_ALIASES.get(normalized) ?? normalized;
|
|
173
|
+
}
|
|
174
|
+
function parseAgentRxFailureCase(value, field) {
|
|
175
|
+
if (typeof value !== "number" && typeof value !== "string") throw new TypeError(`${field} must be a taxonomy number or label`);
|
|
176
|
+
if (typeof value === "string" && !/^\d+$/.test(value.trim())) {
|
|
177
|
+
const normalized = normalizeAgentRxCategory(value);
|
|
178
|
+
const failureCase = AGENT_RX_TAXONOMY_BY_LABEL.get(normalized);
|
|
179
|
+
if (failureCase === void 0) throw new RangeError(`${field} '${value}' is not an AgentRx taxonomy label`);
|
|
180
|
+
return failureCase;
|
|
181
|
+
}
|
|
182
|
+
const numeric = typeof value === "number" ? value : Number(value);
|
|
183
|
+
if (!Number.isSafeInteger(numeric)) throw new TypeError(`${field} must be a taxonomy number or label`);
|
|
184
|
+
if (numeric < 0 || numeric > 10) throw new RangeError(`${field} ${numeric} is outside 0-10`);
|
|
185
|
+
return numeric;
|
|
186
|
+
}
|
|
187
|
+
function parseAgentRxPredictions(output, trajectoryId) {
|
|
188
|
+
let failures;
|
|
189
|
+
let report;
|
|
190
|
+
if (Array.isArray(output)) failures = output;
|
|
191
|
+
else if (isRecord$2(output)) {
|
|
192
|
+
assertMatchingAgentRxTaskId(output.task_id, trajectoryId, "report.task_id");
|
|
193
|
+
if (!Object.hasOwn(output, "failures")) throw new TypeError(`AgentRx prediction '${trajectoryId}' report must contain failures`);
|
|
194
|
+
failures = output.failures;
|
|
195
|
+
if (output.num_judges !== void 0) {
|
|
196
|
+
if (!Number.isSafeInteger(output.num_judges) || output.num_judges < 0) throw new RangeError(`AgentRx prediction '${trajectoryId}' report.num_judges must be a non-negative safe integer`);
|
|
197
|
+
}
|
|
198
|
+
if (output.trajectory_length !== void 0) positiveStep(output.trajectory_length, `AgentRx prediction '${trajectoryId}' report.trajectory_length`);
|
|
199
|
+
if (output.step_mean !== void 0) {
|
|
200
|
+
if (typeof output.step_mean !== "number" || !Number.isFinite(output.step_mean)) throw new TypeError(`AgentRx prediction '${trajectoryId}' report.step_mean must be finite`);
|
|
201
|
+
}
|
|
202
|
+
if (output.modes !== void 0 && !Array.isArray(output.modes)) throw new TypeError(`AgentRx prediction '${trajectoryId}' report.modes must be an array`);
|
|
203
|
+
report = output;
|
|
204
|
+
} else throw new TypeError(`AgentRx prediction '${trajectoryId}' must be a report or failures array`);
|
|
205
|
+
if (!Array.isArray(failures)) throw new TypeError(`AgentRx prediction '${trajectoryId}' failures must be an array`);
|
|
206
|
+
if (failures.length === 0) throw new TypeError(`AgentRx prediction '${trajectoryId}' failures must contain a judge prediction`);
|
|
207
|
+
if (isRecord$2(output) && output.num_judges !== void 0 && output.num_judges !== failures.length) throw new TypeError(`AgentRx prediction '${trajectoryId}' declares ${output.num_judges} judges but contains ${failures.length} failures`);
|
|
208
|
+
return {
|
|
209
|
+
predictions: failures.map((value, index) => {
|
|
210
|
+
const field = `AgentRx prediction '${trajectoryId}' failures[${index}]`;
|
|
211
|
+
if (!isRecord$2(value)) throw new TypeError(`${field} must be an object`);
|
|
212
|
+
assertMatchingAgentRxTaskId(value.task_id, trajectoryId, `${field}.task_id`);
|
|
213
|
+
const failureCase = parseAgentRxFailureCase(value.failure_case, `${field}.failure_case`);
|
|
214
|
+
if (!Number.isSafeInteger(value.step_number)) throw new TypeError(`${field}.step_number must be a safe integer`);
|
|
215
|
+
const stepNumber = value.step_number;
|
|
216
|
+
if (failureCase === 0 ? stepNumber !== 0 : stepNumber < 1) throw new RangeError(failureCase === 0 ? `${field}.step_number must be 0 when failure_case is 0` : `${field}.step_number must be positive when failure_case is 1-10`);
|
|
217
|
+
if (value.description !== void 0 && typeof value.description !== "string") throw new TypeError(`${field}.description must be a string`);
|
|
218
|
+
if (value.checklist_reasoning !== void 0 && value.checklist_reasoning !== null && typeof value.checklist_reasoning !== "string") throw new TypeError(`${field}.checklist_reasoning must be a string or null`);
|
|
219
|
+
return {
|
|
220
|
+
...value.task_id === void 0 ? {} : { task_id: value.task_id },
|
|
221
|
+
failure_case: failureCase,
|
|
222
|
+
step_number: stepNumber,
|
|
223
|
+
...value.description === void 0 ? {} : { description: value.description },
|
|
224
|
+
...value.checklist_reasoning === void 0 ? {} : { checklist_reasoning: value.checklist_reasoning }
|
|
225
|
+
};
|
|
226
|
+
}),
|
|
227
|
+
report
|
|
228
|
+
};
|
|
229
|
+
}
|
|
230
|
+
function agentRxConsensus(parsed, trajectoryId) {
|
|
231
|
+
const counts = /* @__PURE__ */ new Map();
|
|
232
|
+
for (const prediction of parsed.predictions) counts.set(prediction.failure_case, (counts.get(prediction.failure_case) ?? 0) + 1);
|
|
233
|
+
const maxVotes = Math.max(...counts.values());
|
|
234
|
+
let failureCase = [...counts].find(([, count]) => count === maxVotes)[0];
|
|
235
|
+
if (parsed.report?.most_common_failure !== void 0) {
|
|
236
|
+
const declared = parseAgentRxFailureCase(parsed.report.most_common_failure, `AgentRx prediction '${trajectoryId}' report.most_common_failure`);
|
|
237
|
+
if ((counts.get(declared) ?? 0) !== maxVotes) throw new TypeError(`AgentRx prediction '${trajectoryId}' report.most_common_failure disagrees with failures`);
|
|
238
|
+
failureCase = declared;
|
|
239
|
+
}
|
|
240
|
+
if (parsed.report?.modes !== void 0) {
|
|
241
|
+
const declaredModes = parsed.report.modes.map((value, index) => parseAgentRxFailureCase(value, `AgentRx prediction '${trajectoryId}' report.modes[${index}]`));
|
|
242
|
+
if (new Set(declaredModes).size !== declaredModes.length) throw new TypeError(`AgentRx prediction '${trajectoryId}' report.modes contains duplicates`);
|
|
243
|
+
const expectedModes = [...counts].filter(([, count]) => count === maxVotes).map(([value]) => value).sort((left, right) => left - right);
|
|
244
|
+
if ([...new Set(declaredModes)].sort((left, right) => left - right).join(",") !== expectedModes.join(",")) throw new TypeError(`AgentRx prediction '${trajectoryId}' report.modes disagrees with failures`);
|
|
245
|
+
}
|
|
246
|
+
const computedStepMean = parsed.predictions.reduce((sum, prediction) => sum + prediction.step_number, 0) / parsed.predictions.length;
|
|
247
|
+
if (parsed.report?.step_mean !== void 0 && Math.abs(parsed.report.step_mean - computedStepMean) > 1e-12) throw new TypeError(`AgentRx prediction '${trajectoryId}' report.step_mean disagrees with failures`);
|
|
248
|
+
const stepMean = parsed.report?.step_mean ?? computedStepMean;
|
|
249
|
+
const step = failureCase === 0 ? 0 : positiveStep(roundAgentRxStep(stepMean), `AgentRx prediction '${trajectoryId}' consensus step`);
|
|
250
|
+
const representative = parsed.predictions.filter((prediction) => prediction.failure_case === failureCase).sort((left, right) => Math.abs(left.step_number - step) - Math.abs(right.step_number - step))[0] ?? parsed.predictions[0];
|
|
251
|
+
return {
|
|
252
|
+
failureCase,
|
|
253
|
+
step,
|
|
254
|
+
stepMean,
|
|
255
|
+
votes: counts.get(failureCase),
|
|
256
|
+
representative
|
|
257
|
+
};
|
|
258
|
+
}
|
|
259
|
+
/** Match Python's round() behavior used by AgentRx for consensus steps. */
|
|
260
|
+
function roundAgentRxStep(value) {
|
|
261
|
+
if (!Number.isFinite(value)) throw new TypeError("AgentRx step mean must be finite");
|
|
262
|
+
const lower = Math.floor(value);
|
|
263
|
+
const fraction = value - lower;
|
|
264
|
+
if (Math.abs(fraction - .5) <= Number.EPSILON * Math.max(1, Math.abs(value))) return lower % 2 === 0 ? lower : lower + 1;
|
|
265
|
+
return Math.round(value);
|
|
266
|
+
}
|
|
267
|
+
function assertMatchingAgentRxTaskId(value, trajectoryId, field) {
|
|
268
|
+
if (value === void 0) return;
|
|
269
|
+
const taskId = externalId(value, `AgentRx prediction '${trajectoryId}' ${field}`);
|
|
270
|
+
if (taskId !== trajectoryId) throw new TypeError(`AgentRx prediction '${trajectoryId}' ${field} '${taskId}' does not match trajectory id`);
|
|
271
|
+
}
|
|
272
|
+
//#endregion
|
|
273
|
+
//#region src/analyst/benchmark-dataset-codetrace.ts
|
|
274
|
+
function codeTraceBenchCase(row, input, options = {}) {
|
|
275
|
+
const trajectoryId = requiredCodeTraceString(row.traj_id, "CodeTraceBench traj_id");
|
|
276
|
+
const taskName = requiredCodeTraceString(row.task_name, `CodeTraceBench '${trajectoryId}' task_name`);
|
|
277
|
+
const agent = requiredCodeTraceString(row.agent, `CodeTraceBench '${trajectoryId}' agent`);
|
|
278
|
+
const model = requiredCodeTraceString(row.model, `CodeTraceBench '${trajectoryId}' model`);
|
|
279
|
+
const difficulty = optionalCodeTraceString(row.difficulty, `CodeTraceBench '${trajectoryId}' difficulty`);
|
|
280
|
+
const category = optionalCodeTraceString(row.category, `CodeTraceBench '${trajectoryId}' category`);
|
|
281
|
+
const sourceRelpath = optionalSourceRelativePath(row.source_relpath, trajectoryId);
|
|
282
|
+
const solved = codeTraceSolved(row.solved, trajectoryId);
|
|
283
|
+
const tags = parseTags(row.tags, trajectoryId);
|
|
284
|
+
const stepCount = positiveStep(row.step_count, `CodeTraceBench '${trajectoryId}' step_count`);
|
|
285
|
+
const stages = parseCodeTraceStages(row.incorrect_stages, trajectoryId);
|
|
286
|
+
const evidenceKind = options.evidenceKind ?? "span";
|
|
287
|
+
const uri = options.stepUri ?? defaultStepUri;
|
|
288
|
+
const labelSet = codeTraceLabelSet(options.labelSet);
|
|
289
|
+
const labels = /* @__PURE__ */ new Set();
|
|
290
|
+
const expectedIssues = stages.flatMap((stage) => {
|
|
291
|
+
const incorrect = stepIssues("incorrect", stage.incorrect_step_ids ?? []);
|
|
292
|
+
const unuseful = stepIssues("unuseful", stage.unuseful_step_ids ?? []);
|
|
293
|
+
return labelSet === "incorrect-only" ? incorrect : [...incorrect, ...unuseful];
|
|
294
|
+
});
|
|
295
|
+
const labelState = expectedIssues.length > 0 ? "positive" : solved === true ? "trusted-negative" : "unlabeled";
|
|
296
|
+
return {
|
|
297
|
+
id: `codetrace:${trajectoryId}`,
|
|
298
|
+
clusterId: `codetrace-task:${taskName}`,
|
|
299
|
+
labelState,
|
|
300
|
+
input,
|
|
301
|
+
expectedIssues,
|
|
302
|
+
...labelState === "unlabeled" ? {} : { labeledEvidence: expectedIssues.flatMap((issue) => issue.evidence ?? []) },
|
|
303
|
+
tags: [
|
|
304
|
+
"codetracebench",
|
|
305
|
+
agent,
|
|
306
|
+
model,
|
|
307
|
+
...difficulty === void 0 ? [] : [difficulty],
|
|
308
|
+
...category === void 0 ? [] : [category],
|
|
309
|
+
...tags
|
|
310
|
+
],
|
|
311
|
+
metadata: {
|
|
312
|
+
benchmark: "CodeTraceBench",
|
|
313
|
+
trajectoryId,
|
|
314
|
+
taskName,
|
|
315
|
+
agent,
|
|
316
|
+
model,
|
|
317
|
+
solved,
|
|
318
|
+
stepCount,
|
|
319
|
+
labelSet,
|
|
320
|
+
...sourceRelpath === void 0 ? {} : { sourceRelpath },
|
|
321
|
+
...difficulty === void 0 ? {} : { difficulty },
|
|
322
|
+
...category === void 0 ? {} : { category }
|
|
323
|
+
}
|
|
324
|
+
};
|
|
325
|
+
function stepIssues(label, steps) {
|
|
326
|
+
return steps.map((rawStep) => {
|
|
327
|
+
const step = positiveStep(rawStep, `CodeTraceBench '${trajectoryId}' ${label} step`);
|
|
328
|
+
if (step > stepCount) throw new RangeError(`CodeTraceBench '${trajectoryId}' ${label} step ${step} exceeds step_count ${stepCount}`);
|
|
329
|
+
const id = `${label}:${step}`;
|
|
330
|
+
if (labels.has(id)) throw new TypeError(`CodeTraceBench '${trajectoryId}' repeats label '${id}'`);
|
|
331
|
+
labels.add(id);
|
|
332
|
+
return {
|
|
333
|
+
id,
|
|
334
|
+
areas: [label],
|
|
335
|
+
evidence: [{
|
|
336
|
+
kind: evidenceKind,
|
|
337
|
+
uri: uri(trajectoryId, step)
|
|
338
|
+
}]
|
|
339
|
+
};
|
|
340
|
+
});
|
|
341
|
+
}
|
|
342
|
+
}
|
|
343
|
+
/** Translate CodeTracer's `codetracer_labels.json` into shared findings. */
|
|
344
|
+
function codeTracerPredictionsToFindings(trajectoryIdValue, predictions, options = {}) {
|
|
345
|
+
const trajectoryId = nonEmpty$1(trajectoryIdValue, "CodeTracer prediction trajectory id");
|
|
346
|
+
const labels = parseCodeTracerPredictionLabels(predictions, trajectoryId);
|
|
347
|
+
const confidence = predictionConfidence(options.confidence);
|
|
348
|
+
const uri = options.stepUri ?? defaultStepUri;
|
|
349
|
+
const labelSet = codeTraceLabelSet(options.labelSet);
|
|
350
|
+
const seen = /* @__PURE__ */ new Set();
|
|
351
|
+
const findings = [];
|
|
352
|
+
for (const label of labels) {
|
|
353
|
+
const step = positiveStep(label.step, `CodeTracer prediction '${trajectoryId}' ${label.area} step`);
|
|
354
|
+
assertStepWithinRange(step, options.stepCount, `CodeTracer prediction '${trajectoryId}'`);
|
|
355
|
+
const key = `${label.area}:${step}`;
|
|
356
|
+
if (seen.has(key)) throw new TypeError(`CodeTracer prediction '${trajectoryId}' repeats label '${key}'`);
|
|
357
|
+
seen.add(key);
|
|
358
|
+
if (label.area === "unuseful" && labelSet === "incorrect-only") continue;
|
|
359
|
+
findings.push(makeFinding({
|
|
360
|
+
analyst_id: options.analystId ?? "codetracer",
|
|
361
|
+
produced_at: options.producedAt,
|
|
362
|
+
area: label.area,
|
|
363
|
+
subject: `step-${step}`,
|
|
364
|
+
claim: `CodeTracer labeled step ${step} as ${label.area}.`,
|
|
365
|
+
id_basis: key,
|
|
366
|
+
rationale: label.reasoning,
|
|
367
|
+
severity: "medium",
|
|
368
|
+
confidence,
|
|
369
|
+
evidence_refs: [{
|
|
370
|
+
kind: options.evidenceKind ?? "span",
|
|
371
|
+
uri: uri(trajectoryId, step)
|
|
372
|
+
}],
|
|
373
|
+
metadata: {
|
|
374
|
+
upstream: "CodeTracer",
|
|
375
|
+
stage_id: label.stageId,
|
|
376
|
+
step
|
|
377
|
+
}
|
|
378
|
+
}));
|
|
379
|
+
}
|
|
380
|
+
return findings;
|
|
381
|
+
}
|
|
382
|
+
function parseCodeTraceStages(value, trajectoryId) {
|
|
383
|
+
let parsed = value;
|
|
384
|
+
if (typeof value === "string") try {
|
|
385
|
+
parsed = JSON.parse(value);
|
|
386
|
+
} catch (error) {
|
|
387
|
+
throw new TypeError(`CodeTraceBench '${trajectoryId}' incorrect_stages is invalid JSON: ${error instanceof Error ? error.message : String(error)}`);
|
|
388
|
+
}
|
|
389
|
+
if (!Array.isArray(parsed)) throw new TypeError(`CodeTraceBench '${trajectoryId}' incorrect_stages must be an array`);
|
|
390
|
+
const stageIds = /* @__PURE__ */ new Set();
|
|
391
|
+
for (const stage of parsed) {
|
|
392
|
+
if (!stage || typeof stage !== "object" || !Number.isSafeInteger(stage.stage_id) || stage.stage_id < 1 || !optionalStepArray(stage.incorrect_step_ids) || !optionalStepArray(stage.unuseful_step_ids) || stage.reasoning !== void 0 && typeof stage.reasoning !== "string") throw new TypeError(`CodeTraceBench '${trajectoryId}' contains an invalid stage annotation`);
|
|
393
|
+
const stageId = stage.stage_id;
|
|
394
|
+
if (stageIds.has(stageId)) throw new TypeError(`CodeTraceBench '${trajectoryId}' repeats stage_id ${stageId}`);
|
|
395
|
+
stageIds.add(stageId);
|
|
396
|
+
}
|
|
397
|
+
return parsed;
|
|
398
|
+
}
|
|
399
|
+
function parseCodeTracerPredictionLabels(value, trajectoryId) {
|
|
400
|
+
let parsed = value;
|
|
401
|
+
if (typeof value === "string") try {
|
|
402
|
+
parsed = JSON.parse(value);
|
|
403
|
+
} catch (error) {
|
|
404
|
+
throw new TypeError(`CodeTracer prediction '${trajectoryId}' is invalid JSON: ${error instanceof Error ? error.message : String(error)}`);
|
|
405
|
+
}
|
|
406
|
+
if (!Array.isArray(parsed)) throw new TypeError(`CodeTracer prediction '${trajectoryId}' must be an array`);
|
|
407
|
+
if (parsed.length === 0) return [];
|
|
408
|
+
if (parsed.every(isCodeTraceStageAnnotation)) return parsed.flatMap((stage) => [...(stage.incorrect_step_ids ?? []).map((step) => ({
|
|
409
|
+
stageId: stage.stage_id,
|
|
410
|
+
area: "incorrect",
|
|
411
|
+
step,
|
|
412
|
+
...stage.reasoning === void 0 ? {} : { reasoning: stage.reasoning }
|
|
413
|
+
})), ...(stage.unuseful_step_ids ?? []).map((step) => ({
|
|
414
|
+
stageId: stage.stage_id,
|
|
415
|
+
area: "unuseful",
|
|
416
|
+
step,
|
|
417
|
+
...stage.reasoning === void 0 ? {} : { reasoning: stage.reasoning }
|
|
418
|
+
}))]);
|
|
419
|
+
const flat = [];
|
|
420
|
+
for (const [index, item] of parsed.entries()) {
|
|
421
|
+
if (isCodeTracerStepLabel(item)) {
|
|
422
|
+
flat.push(toCodeTracerPredictionLabel(item, codeTracerStageId(item, index + 1, trajectoryId), trajectoryId));
|
|
423
|
+
continue;
|
|
424
|
+
}
|
|
425
|
+
if (!isRecord$2(item) || !Array.isArray(item.labels)) throw new TypeError(`CodeTracer prediction '${trajectoryId}' contains an unsupported label row`);
|
|
426
|
+
const stageId = codeTracerStageId(item, index + 1, trajectoryId);
|
|
427
|
+
for (const label of item.labels) {
|
|
428
|
+
if (!isCodeTracerStepLabel(label)) throw new TypeError(`CodeTracer prediction '${trajectoryId}' contains an invalid step label`);
|
|
429
|
+
flat.push(toCodeTracerPredictionLabel(label, stageId, trajectoryId));
|
|
430
|
+
}
|
|
431
|
+
}
|
|
432
|
+
return flat;
|
|
433
|
+
}
|
|
434
|
+
function isCodeTraceStageAnnotation(value) {
|
|
435
|
+
return isRecord$2(value) && Number.isSafeInteger(value.stage_id) && value.stage_id > 0 && optionalStepArray(value.incorrect_step_ids) && optionalStepArray(value.unuseful_step_ids) && (value.reasoning === void 0 || typeof value.reasoning === "string");
|
|
436
|
+
}
|
|
437
|
+
function isCodeTracerStepLabel(value) {
|
|
438
|
+
return isRecord$2(value) && Number.isSafeInteger(value.step_id) && value.step_id > 0 && (value.stage === void 0 || typeof value.stage === "string" || typeof value.stage === "number") && (value.stage_name === void 0 || typeof value.stage_name === "string" || typeof value.stage_name === "number") && (value.stage_id === void 0 || typeof value.stage_id === "string" || typeof value.stage_id === "number") && (value.label === "incorrect" || value.label === "unuseful") && (value.rationale === void 0 || typeof value.rationale === "string") && (value.reason === void 0 || typeof value.reason === "string") && (value.note === void 0 || typeof value.note === "string") && (value.comment === void 0 || typeof value.comment === "string");
|
|
439
|
+
}
|
|
440
|
+
function toCodeTracerPredictionLabel(label, stageId, trajectoryId) {
|
|
441
|
+
if (!isCodeTracerStepLabel(label)) throw new TypeError(`CodeTracer prediction '${trajectoryId}' contains an invalid step label`);
|
|
442
|
+
return {
|
|
443
|
+
stageId,
|
|
444
|
+
area: label.label,
|
|
445
|
+
step: label.step_id,
|
|
446
|
+
...codeTracerReason(label, trajectoryId)
|
|
447
|
+
};
|
|
448
|
+
}
|
|
449
|
+
function codeTracerStageId(value, fallback, trajectoryId) {
|
|
450
|
+
const candidates = [
|
|
451
|
+
value.stage,
|
|
452
|
+
value.stage_name,
|
|
453
|
+
value.stage_id
|
|
454
|
+
].filter((candidate) => typeof candidate === "string" || typeof candidate === "number");
|
|
455
|
+
if (new Set(candidates).size > 1) throw new TypeError(`CodeTracer prediction '${trajectoryId}' has conflicting stage fields`);
|
|
456
|
+
return candidates[0] ?? fallback;
|
|
457
|
+
}
|
|
458
|
+
function codeTracerReason(label, trajectoryId) {
|
|
459
|
+
const candidates = [
|
|
460
|
+
label.rationale,
|
|
461
|
+
label.reason,
|
|
462
|
+
label.note,
|
|
463
|
+
label.comment
|
|
464
|
+
].filter((candidate) => candidate !== void 0);
|
|
465
|
+
if (new Set(candidates).size > 1) throw new TypeError(`CodeTracer prediction '${trajectoryId}' step ${label.step_id} has conflicting reason fields`);
|
|
466
|
+
return candidates[0] === void 0 ? {} : { reasoning: candidates[0] };
|
|
467
|
+
}
|
|
468
|
+
function parseTags(value, trajectoryId) {
|
|
469
|
+
if (value === void 0) return [];
|
|
470
|
+
let parsed = value;
|
|
471
|
+
if (typeof value === "string") try {
|
|
472
|
+
parsed = JSON.parse(value);
|
|
473
|
+
} catch (error) {
|
|
474
|
+
throw new TypeError(`CodeTraceBench '${trajectoryId}' tags is invalid JSON: ${error instanceof Error ? error.message : String(error)}`);
|
|
475
|
+
}
|
|
476
|
+
if (!Array.isArray(parsed)) throw new TypeError(`CodeTraceBench '${trajectoryId}' tags must be an array of strings`);
|
|
477
|
+
const tags = parsed.map((tag, index) => requiredCodeTraceString(tag, `CodeTraceBench '${trajectoryId}' tags[${index}]`));
|
|
478
|
+
if (new Set(tags).size !== tags.length) throw new TypeError(`CodeTraceBench '${trajectoryId}' tags must not repeat values`);
|
|
479
|
+
return tags;
|
|
480
|
+
}
|
|
481
|
+
function codeTraceLabelSet(value) {
|
|
482
|
+
if (value === void 0 || value === "incorrect-only") return "incorrect-only";
|
|
483
|
+
if (value === "incorrect-and-unuseful") return value;
|
|
484
|
+
throw new TypeError("CodeTraceBench labelSet must be 'incorrect-only' or 'incorrect-and-unuseful'");
|
|
485
|
+
}
|
|
486
|
+
function optionalStepArray(value) {
|
|
487
|
+
return value === void 0 || Array.isArray(value) && value.every(Number.isSafeInteger);
|
|
488
|
+
}
|
|
489
|
+
function requiredCodeTraceString(value, field) {
|
|
490
|
+
if (typeof value !== "string") throw new TypeError(`${field} must be a string`);
|
|
491
|
+
return nonEmpty$1(value, field);
|
|
492
|
+
}
|
|
493
|
+
function optionalCodeTraceString(value, field) {
|
|
494
|
+
if (value === void 0) return void 0;
|
|
495
|
+
return requiredCodeTraceString(value, field);
|
|
496
|
+
}
|
|
497
|
+
function optionalSourceRelativePath(value, trajectoryId) {
|
|
498
|
+
const path = optionalCodeTraceString(value, `CodeTraceBench '${trajectoryId}' source_relpath`);
|
|
499
|
+
if (path === void 0) return void 0;
|
|
500
|
+
const segments = path.replaceAll("\\", "/").split("/");
|
|
501
|
+
if (path.startsWith("/") || /^[a-zA-Z]:[\\/]/.test(path) || segments.some((segment) => segment === "..")) throw new TypeError(`CodeTraceBench '${trajectoryId}' source_relpath must stay within the artifact root`);
|
|
502
|
+
return path;
|
|
503
|
+
}
|
|
504
|
+
function codeTraceSolved(value, trajectoryId) {
|
|
505
|
+
if (value === void 0 || value === null || typeof value === "boolean") return value;
|
|
506
|
+
throw new TypeError(`CodeTraceBench '${trajectoryId}' solved must be a boolean or null`);
|
|
507
|
+
}
|
|
508
|
+
//#endregion
|
|
509
|
+
//#region src/analyst/benchmark-agentrx-calibration.ts
|
|
510
|
+
const AGENT_RX_UPSTREAM_REVISION = "f228165bfec60a801fd5fedd9d8ffe0f9de0c69d";
|
|
511
|
+
function summarizeAgentRxCalibration(result, upstreamRevision) {
|
|
512
|
+
if (!upstreamRevision.trim()) throw new TypeError("AgentRx calibration requires an upstream revision");
|
|
513
|
+
return {
|
|
514
|
+
protocol: "official-agentrx-root-cause",
|
|
515
|
+
upstreamRevision,
|
|
516
|
+
rationale: "Matches AgentRx root-category accuracy, Python-rounded exact and tolerance step accuracy, unrounded mean step distance, normalized distance, and any, earliest, and terminal category accuracy. Failed runs and empty predictions score as no prediction.",
|
|
517
|
+
runners: result.provenance.runnerIds.map((runnerId) => summarizeRunner$1(runnerId, result.observations.filter((observation) => observation.runnerId === runnerId)))
|
|
518
|
+
};
|
|
519
|
+
}
|
|
520
|
+
function renderAgentRxCalibrationMarkdown(summary) {
|
|
521
|
+
return [
|
|
522
|
+
"## AgentRx Published Metrics",
|
|
523
|
+
"",
|
|
524
|
+
summary.rationale,
|
|
525
|
+
"",
|
|
526
|
+
`Upstream revision: \`${summary.upstreamRevision}\`.`,
|
|
527
|
+
"",
|
|
528
|
+
"| Runner | Completed/selected | Failed | Predictions | Missing predictions | Exact step | Within 1 | Within 2 | Within 3 | Within 4 | Within 5 | Mean step distance | Normalized distance | Normalized known/unknown | Root category | Any category | Earliest category | Terminal category |",
|
|
529
|
+
"| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |",
|
|
530
|
+
...summary.runners.map((runner) => `| ${escapeCell$2(runner.runnerId)} | ${runner.completedRuns}/${runner.selectedRuns} | ${runner.failedRuns} | ${runner.predictedRuns} | ${runner.missingPredictionRuns} | ${rate$3(runner.exactStepAccuracy)} | ${rate$3(runner.stepAccuracyWithin1)} | ${rate$3(runner.stepAccuracyWithin2)} | ${rate$3(runner.stepAccuracyWithin3)} | ${rate$3(runner.stepAccuracyWithin4)} | ${rate$3(runner.stepAccuracyWithin5)} | ${number$1(runner.meanStepDistance)} | ${number$1(runner.normalizedMeanStepDistance)} | ${runner.normalizedDistanceRuns}/${runner.normalizedDistanceUnknownRuns} | ${rate$3(runner.rootCauseCategoryAccuracy)} | ${rate$3(runner.anyFailureCategoryAccuracy)} | ${rate$3(runner.earliestFailureCategoryAccuracy)} | ${rate$3(runner.terminalFailureCategoryAccuracy)} |`)
|
|
531
|
+
].join("\n");
|
|
532
|
+
}
|
|
533
|
+
function summarizeRunner$1(runnerId, observations) {
|
|
534
|
+
const scored = observations.map(scoredObservation);
|
|
535
|
+
const normalized = scored.filter((row) => row.normalizedDistance !== null);
|
|
536
|
+
const predicted = scored.filter((row) => row.distance !== null);
|
|
537
|
+
return {
|
|
538
|
+
runnerId,
|
|
539
|
+
selectedRuns: observations.length,
|
|
540
|
+
completedRuns: observations.filter((observation) => !observation.error).length,
|
|
541
|
+
failedRuns: observations.filter((observation) => Boolean(observation.error)).length,
|
|
542
|
+
predictedRuns: predicted.length,
|
|
543
|
+
missingPredictionRuns: scored.length - predicted.length,
|
|
544
|
+
exactStepAccuracy: mean$2(scored.map((row) => Number(row.roundedDistance === 0))),
|
|
545
|
+
stepAccuracyWithin1: mean$2(scored.map((row) => Number(row.roundedDistance !== null && row.roundedDistance <= 1))),
|
|
546
|
+
stepAccuracyWithin2: mean$2(scored.map((row) => Number(row.roundedDistance !== null && row.roundedDistance <= 2))),
|
|
547
|
+
stepAccuracyWithin3: mean$2(scored.map((row) => Number(row.roundedDistance !== null && row.roundedDistance <= 3))),
|
|
548
|
+
stepAccuracyWithin4: mean$2(scored.map((row) => Number(row.roundedDistance !== null && row.roundedDistance <= 4))),
|
|
549
|
+
stepAccuracyWithin5: mean$2(scored.map((row) => Number(row.roundedDistance !== null && row.roundedDistance <= 5))),
|
|
550
|
+
meanStepDistance: mean$2(predicted.map((row) => row.distance)),
|
|
551
|
+
normalizedMeanStepDistance: mean$2(normalized.map((row) => row.normalizedDistance)),
|
|
552
|
+
normalizedDistanceRuns: normalized.length,
|
|
553
|
+
normalizedDistanceUnknownRuns: scored.length - normalized.length,
|
|
554
|
+
rootCauseCategoryAccuracy: mean$2(scored.map((row) => Number(row.rootCategoryMatch))),
|
|
555
|
+
anyFailureCategoryAccuracy: mean$2(scored.map((row) => Number(row.anyCategoryMatch))),
|
|
556
|
+
earliestFailureCategoryAccuracy: mean$2(scored.map((row) => Number(row.earliestCategoryMatch))),
|
|
557
|
+
terminalFailureCategoryAccuracy: mean$2(scored.map((row) => Number(row.terminalCategoryMatch)))
|
|
558
|
+
};
|
|
559
|
+
}
|
|
560
|
+
function scoredObservation(observation) {
|
|
561
|
+
const metadata = record(observation.caseMetadata);
|
|
562
|
+
const rootStep = requiredPositiveNumber(metadata.rootCauseStep, observation.caseId, "rootCauseStep");
|
|
563
|
+
const rootCategory = requiredString$1(metadata.rootCauseCategory, observation.caseId, "rootCauseCategory");
|
|
564
|
+
const allCategories = requiredStringArray(metadata.allFailureCategories, observation.caseId, "allFailureCategories");
|
|
565
|
+
const earliestCategory = requiredString$1(metadata.earliestFailureCategory, observation.caseId, "earliestFailureCategory");
|
|
566
|
+
const terminalCategory = requiredString$1(metadata.terminalFailureCategory, observation.caseId, "terminalFailureCategory");
|
|
567
|
+
const finding = observation.error ? void 0 : observation.findings[0];
|
|
568
|
+
const findingMetadata = record(finding?.metadata);
|
|
569
|
+
const stepMean = finding ? finiteNonNegative(findingMetadata.step_mean ?? findingMetadata.step, observation.caseId, "predicted step") : null;
|
|
570
|
+
const roundedDistance = stepMean === null ? null : Math.abs(roundAgentRxStep(stepMean) - rootStep);
|
|
571
|
+
const distance = stepMean === null ? null : Math.abs(stepMean - rootStep);
|
|
572
|
+
const trajectoryLength = metadata.trajectoryLength === void 0 ? null : requiredPositiveNumber(metadata.trajectoryLength, observation.caseId, "trajectoryLength");
|
|
573
|
+
const predictedCategory = finding?.area;
|
|
574
|
+
return {
|
|
575
|
+
roundedDistance,
|
|
576
|
+
distance,
|
|
577
|
+
normalizedDistance: trajectoryLength === null || distance === null ? null : distance / trajectoryLength,
|
|
578
|
+
rootCategoryMatch: predictedCategory === rootCategory,
|
|
579
|
+
anyCategoryMatch: predictedCategory !== void 0 && allCategories.includes(predictedCategory),
|
|
580
|
+
earliestCategoryMatch: predictedCategory === earliestCategory,
|
|
581
|
+
terminalCategoryMatch: predictedCategory === terminalCategory
|
|
582
|
+
};
|
|
583
|
+
}
|
|
584
|
+
function record(value) {
|
|
585
|
+
return typeof value === "object" && value !== null && !Array.isArray(value) ? value : {};
|
|
586
|
+
}
|
|
587
|
+
function requiredPositiveNumber(value, caseId, field) {
|
|
588
|
+
const numberValue = finiteNonNegative(value, caseId, field);
|
|
589
|
+
if (numberValue <= 0) throw new TypeError(`${caseId}: ${field} must be positive`);
|
|
590
|
+
return numberValue;
|
|
591
|
+
}
|
|
592
|
+
function finiteNonNegative(value, caseId, field) {
|
|
593
|
+
if (typeof value !== "number" || !Number.isFinite(value) || value < 0) throw new TypeError(`${caseId}: ${field} must be a finite non-negative number`);
|
|
594
|
+
return value;
|
|
595
|
+
}
|
|
596
|
+
function requiredString$1(value, caseId, field) {
|
|
597
|
+
if (typeof value !== "string" || !value.trim()) throw new TypeError(`${caseId}: ${field} must be a non-empty string`);
|
|
598
|
+
return value;
|
|
599
|
+
}
|
|
600
|
+
function requiredStringArray(value, caseId, field) {
|
|
601
|
+
if (!Array.isArray(value) || value.length === 0 || value.some((entry) => typeof entry !== "string" || !entry.trim())) throw new TypeError(`${caseId}: ${field} must be a non-empty string array`);
|
|
602
|
+
return value;
|
|
603
|
+
}
|
|
604
|
+
function mean$2(values) {
|
|
605
|
+
return values.length === 0 ? null : values.reduce((total, value) => total + value, 0) / values.length;
|
|
606
|
+
}
|
|
607
|
+
function rate$3(value) {
|
|
608
|
+
return value === null ? "n/a" : value.toFixed(3);
|
|
609
|
+
}
|
|
610
|
+
function number$1(value) {
|
|
611
|
+
return value === null ? "n/a" : value.toFixed(3);
|
|
612
|
+
}
|
|
613
|
+
function escapeCell$2(value) {
|
|
614
|
+
return value.replaceAll("|", "\\|").replaceAll("\n", " ");
|
|
615
|
+
}
|
|
616
|
+
//#endregion
|
|
617
|
+
//#region src/analyst/benchmark-command-validation.ts
|
|
618
|
+
const nonEmptyString = z.string().refine((value) => value.trim().length > 0, { message: "must be a non-empty string" });
|
|
619
|
+
const safeInteger$1 = z.number().refine(Number.isSafeInteger, { message: "must be a safe integer" });
|
|
620
|
+
const nonNegativeInteger = safeInteger$1.refine((value) => value >= 0, { message: "must be a non-negative safe integer" });
|
|
621
|
+
const positiveInteger$1 = safeInteger$1.refine((value) => value > 0, { message: "must be a positive safe integer" });
|
|
622
|
+
const nonNegativeNumber = z.number().nonnegative();
|
|
623
|
+
const rate$2 = z.number().min(0).max(1);
|
|
624
|
+
const nullableRate = rate$2.nullable();
|
|
625
|
+
const nullableFiniteNumber = z.number().nullable();
|
|
626
|
+
const sha256 = z.string().regex(/^[a-f0-9]{64}$/, "must be a lowercase SHA-256 digest");
|
|
627
|
+
const revision = z.string().regex(/^(?:[a-f0-9]{40}|[a-f0-9]{64})$/, "must be a lowercase 40 or 64 character revision");
|
|
628
|
+
const timestamp = z.string().refine((value) => Number.isFinite(Date.parse(value)), { message: "must be a valid timestamp" });
|
|
629
|
+
const stringArray = z.array(z.string());
|
|
630
|
+
const nonEmptyStringArray = z.array(nonEmptyString);
|
|
631
|
+
const metadata = z.record(z.string(), z.unknown());
|
|
632
|
+
const errorSchema = z.strictObject({
|
|
633
|
+
class: nonEmptyString,
|
|
634
|
+
message: nonEmptyString,
|
|
635
|
+
code: nonEmptyString.optional(),
|
|
636
|
+
status: z.number().int().min(100).max(599).optional()
|
|
637
|
+
});
|
|
638
|
+
const evidenceSchema = z.strictObject({
|
|
639
|
+
kind: z.enum([
|
|
640
|
+
"span",
|
|
641
|
+
"event",
|
|
642
|
+
"artifact",
|
|
643
|
+
"finding",
|
|
644
|
+
"metric"
|
|
645
|
+
]),
|
|
646
|
+
uri: nonEmptyString,
|
|
647
|
+
excerpt: z.string().optional()
|
|
648
|
+
});
|
|
649
|
+
const findingSchema = z.strictObject({
|
|
650
|
+
schema_version: z.literal("1.0.0"),
|
|
651
|
+
finding_id: nonEmptyString,
|
|
652
|
+
analyst_id: nonEmptyString,
|
|
653
|
+
produced_at: timestamp,
|
|
654
|
+
severity: z.enum([
|
|
655
|
+
"critical",
|
|
656
|
+
"high",
|
|
657
|
+
"medium",
|
|
658
|
+
"low",
|
|
659
|
+
"info"
|
|
660
|
+
]),
|
|
661
|
+
area: nonEmptyString,
|
|
662
|
+
claim: nonEmptyString,
|
|
663
|
+
rationale: z.string().optional(),
|
|
664
|
+
evidence_refs: z.array(evidenceSchema),
|
|
665
|
+
recommended_action: z.string().optional(),
|
|
666
|
+
validation_plan: z.string().optional(),
|
|
667
|
+
confidence: rate$2,
|
|
668
|
+
subject: z.string().optional(),
|
|
669
|
+
derived_from_judge: z.boolean().optional(),
|
|
670
|
+
metadata: metadata.optional()
|
|
671
|
+
});
|
|
672
|
+
const tokenUsageSchema = z.strictObject({
|
|
673
|
+
input: nonNegativeInteger,
|
|
674
|
+
output: nonNegativeInteger,
|
|
675
|
+
reasoning: nonNegativeInteger.optional(),
|
|
676
|
+
cached: nonNegativeInteger.optional(),
|
|
677
|
+
cacheWrite: nonNegativeInteger.optional()
|
|
678
|
+
}).superRefine((usage, context) => {
|
|
679
|
+
if (usage.reasoning !== void 0 && usage.reasoning > usage.output) context.addIssue({
|
|
680
|
+
code: "custom",
|
|
681
|
+
path: ["reasoning"],
|
|
682
|
+
message: "must not exceed output tokens"
|
|
683
|
+
});
|
|
684
|
+
});
|
|
685
|
+
const costSchema = z.discriminatedUnion("kind", [
|
|
686
|
+
z.strictObject({
|
|
687
|
+
kind: z.literal("observed"),
|
|
688
|
+
usd: nonNegativeNumber
|
|
689
|
+
}),
|
|
690
|
+
z.strictObject({
|
|
691
|
+
kind: z.literal("estimated"),
|
|
692
|
+
usd: nonNegativeNumber
|
|
693
|
+
}),
|
|
694
|
+
z.strictObject({
|
|
695
|
+
kind: z.literal("uncaptured"),
|
|
696
|
+
usd: z.null()
|
|
697
|
+
})
|
|
698
|
+
]);
|
|
699
|
+
const usageSchema = z.strictObject({
|
|
700
|
+
calls: nonNegativeInteger.nullable(),
|
|
701
|
+
tokens: tokenUsageSchema.nullable(),
|
|
702
|
+
cost: costSchema,
|
|
703
|
+
knownCostUsd: nonNegativeNumber.optional()
|
|
704
|
+
});
|
|
705
|
+
const findingScoreSchema = z.strictObject({
|
|
706
|
+
expectedIssueCount: nonNegativeInteger,
|
|
707
|
+
matchedIssueIds: nonEmptyStringArray,
|
|
708
|
+
missedIssueIds: nonEmptyStringArray,
|
|
709
|
+
supportedFindingIndexes: z.array(nonNegativeInteger),
|
|
710
|
+
unsupportedFindingIndexes: z.array(nonNegativeInteger),
|
|
711
|
+
unlabeledEvidence: z.array(evidenceSchema),
|
|
712
|
+
issueRecall: rate$2,
|
|
713
|
+
findingPrecision: rate$2,
|
|
714
|
+
f1: rate$2,
|
|
715
|
+
criticalStepAccuracy: nullableRate,
|
|
716
|
+
citationCoverage: nullableRate,
|
|
717
|
+
citationExcerptCoverage: nullableRate,
|
|
718
|
+
citationLabelAgreement: nullableRate,
|
|
719
|
+
predictionOnLabelEmptyCase: z.boolean()
|
|
720
|
+
});
|
|
721
|
+
const evidenceResolutionSchema = z.strictObject({
|
|
722
|
+
checked: nonNegativeInteger,
|
|
723
|
+
resolved: nonNegativeInteger,
|
|
724
|
+
unresolvedEvidence: z.array(evidenceSchema),
|
|
725
|
+
errors: z.array(z.strictObject({
|
|
726
|
+
evidence: evidenceSchema,
|
|
727
|
+
class: nonEmptyString,
|
|
728
|
+
message: nonEmptyString
|
|
729
|
+
})),
|
|
730
|
+
validity: nullableRate
|
|
731
|
+
});
|
|
732
|
+
const observationSchema = z.strictObject({
|
|
733
|
+
runnerId: nonEmptyString,
|
|
734
|
+
caseId: nonEmptyString,
|
|
735
|
+
clusterId: nonEmptyString,
|
|
736
|
+
labelState: z.enum([
|
|
737
|
+
"positive",
|
|
738
|
+
"trusted-negative",
|
|
739
|
+
"unlabeled"
|
|
740
|
+
]),
|
|
741
|
+
repetition: nonNegativeInteger,
|
|
742
|
+
executionIndex: nonNegativeInteger,
|
|
743
|
+
latencyMs: nonNegativeNumber.nullable(),
|
|
744
|
+
latencySource: z.enum([
|
|
745
|
+
"benchmark-clock",
|
|
746
|
+
"runner-reported",
|
|
747
|
+
"uncaptured"
|
|
748
|
+
]),
|
|
749
|
+
findings: z.array(findingSchema),
|
|
750
|
+
score: findingScoreSchema,
|
|
751
|
+
evidenceResolution: evidenceResolutionSchema.optional(),
|
|
752
|
+
caseTags: stringArray,
|
|
753
|
+
caseMetadata: metadata.optional(),
|
|
754
|
+
usage: usageSchema.optional(),
|
|
755
|
+
runnerMetadata: metadata.optional(),
|
|
756
|
+
error: errorSchema.optional()
|
|
757
|
+
}).superRefine((observation, context) => {
|
|
758
|
+
const latencyIsMissing = observation.latencyMs === null;
|
|
759
|
+
if (observation.latencySource === "uncaptured" && !latencyIsMissing || observation.latencySource !== "uncaptured" && latencyIsMissing) context.addIssue({
|
|
760
|
+
code: "custom",
|
|
761
|
+
path: ["latencyMs"],
|
|
762
|
+
message: `must ${observation.latencySource === "uncaptured" ? "" : "not "}be null for '${observation.latencySource}' latency`
|
|
763
|
+
});
|
|
764
|
+
});
|
|
765
|
+
const latencyDistributionSchema = z.strictObject({
|
|
766
|
+
min: nonNegativeNumber,
|
|
767
|
+
mean: nonNegativeNumber,
|
|
768
|
+
p50: nonNegativeNumber,
|
|
769
|
+
p95: nonNegativeNumber,
|
|
770
|
+
max: nonNegativeNumber
|
|
771
|
+
});
|
|
772
|
+
const summarySchema = z.strictObject({
|
|
773
|
+
runnerId: nonEmptyString,
|
|
774
|
+
plannedRuns: nonNegativeInteger,
|
|
775
|
+
completedRuns: nonNegativeInteger,
|
|
776
|
+
failedRuns: nonNegativeInteger,
|
|
777
|
+
issueBearingRuns: nonNegativeInteger,
|
|
778
|
+
trustedNegativeRuns: nonNegativeInteger,
|
|
779
|
+
unlabeledRuns: nonNegativeInteger,
|
|
780
|
+
issueRecall: nullableRate,
|
|
781
|
+
findingPrecision: nullableRate,
|
|
782
|
+
f1: nullableRate,
|
|
783
|
+
macroIssueRecall: nullableRate,
|
|
784
|
+
macroFindingPrecision: nullableRate,
|
|
785
|
+
macroF1: nullableRate,
|
|
786
|
+
criticalStepAccuracy: nullableRate,
|
|
787
|
+
citationCoverage: nullableRate,
|
|
788
|
+
citationExcerptCoverage: nullableRate,
|
|
789
|
+
citationLabelAgreement: nullableRate,
|
|
790
|
+
citationResolution: nullableRate,
|
|
791
|
+
citationResolutionUnknownRuns: nonNegativeInteger,
|
|
792
|
+
unresolvedCitations: nonNegativeInteger,
|
|
793
|
+
citationResolutionErrors: nonNegativeInteger,
|
|
794
|
+
trustedNegativeFalsePositiveRate: nullableRate,
|
|
795
|
+
trustedNegativeFailureRate: nullableRate,
|
|
796
|
+
unlabeledPredictionRate: nullableRate,
|
|
797
|
+
unlabeledFailureRate: nullableRate,
|
|
798
|
+
predictionAgreement: nullableRate,
|
|
799
|
+
predictionAgreementCases: nonNegativeInteger,
|
|
800
|
+
matchedLabelAgreement: nullableRate,
|
|
801
|
+
matchedLabelAgreementCases: nonNegativeInteger,
|
|
802
|
+
latencyMs: latencyDistributionSchema.nullable(),
|
|
803
|
+
benchmarkClockLatencyRuns: nonNegativeInteger,
|
|
804
|
+
runnerReportedLatencyRuns: nonNegativeInteger,
|
|
805
|
+
latencyUnknownRuns: nonNegativeInteger,
|
|
806
|
+
calls: nonNegativeInteger,
|
|
807
|
+
callsUnknownRuns: nonNegativeInteger,
|
|
808
|
+
inputTokens: nonNegativeInteger,
|
|
809
|
+
outputTokens: nonNegativeInteger,
|
|
810
|
+
reasoningTokens: nonNegativeInteger,
|
|
811
|
+
cachedTokens: nonNegativeInteger,
|
|
812
|
+
cacheWriteTokens: nonNegativeInteger,
|
|
813
|
+
tokenUsageUnknownRuns: nonNegativeInteger,
|
|
814
|
+
reasoningTokenUsageUnknownRuns: nonNegativeInteger,
|
|
815
|
+
cachedTokenUsageUnknownRuns: nonNegativeInteger,
|
|
816
|
+
cacheWriteTokenUsageUnknownRuns: nonNegativeInteger,
|
|
817
|
+
knownCostUsd: nonNegativeNumber,
|
|
818
|
+
costUnknownRuns: nonNegativeInteger
|
|
819
|
+
});
|
|
820
|
+
const provenanceSchema = z.strictObject({
|
|
821
|
+
id: nonEmptyString.optional(),
|
|
822
|
+
dataset: z.strictObject({
|
|
823
|
+
id: nonEmptyString,
|
|
824
|
+
revision: nonEmptyString,
|
|
825
|
+
split: nonEmptyString.optional()
|
|
826
|
+
}).optional(),
|
|
827
|
+
command: nonEmptyString.optional(),
|
|
828
|
+
environment: z.record(z.string(), z.string()).optional(),
|
|
829
|
+
metadata: metadata.optional(),
|
|
830
|
+
startedAt: timestamp,
|
|
831
|
+
endedAt: timestamp,
|
|
832
|
+
caseCount: positiveInteger$1,
|
|
833
|
+
runnerIds: nonEmptyStringArray.min(1),
|
|
834
|
+
repetitions: positiveInteger$1,
|
|
835
|
+
maxConcurrency: positiveInteger$1,
|
|
836
|
+
runnerOrderSeed: safeInteger$1
|
|
837
|
+
}).superRefine((provenance, context) => {
|
|
838
|
+
if (Date.parse(provenance.endedAt) < Date.parse(provenance.startedAt)) context.addIssue({
|
|
839
|
+
code: "custom",
|
|
840
|
+
path: ["endedAt"],
|
|
841
|
+
message: "must not precede startedAt"
|
|
842
|
+
});
|
|
843
|
+
});
|
|
844
|
+
const resultSchema = z.strictObject({
|
|
845
|
+
provenance: provenanceSchema,
|
|
846
|
+
observations: z.array(observationSchema),
|
|
847
|
+
summaries: z.array(summarySchema)
|
|
848
|
+
});
|
|
849
|
+
const comparisonMetricSchema = z.strictObject({
|
|
850
|
+
metric: z.enum([
|
|
851
|
+
"completion",
|
|
852
|
+
"issueRecall",
|
|
853
|
+
"findingPrecision",
|
|
854
|
+
"f1",
|
|
855
|
+
"criticalStepAccuracy",
|
|
856
|
+
"citationCoverage",
|
|
857
|
+
"citationExcerptCoverage",
|
|
858
|
+
"citationLabelAgreement",
|
|
859
|
+
"citationResolution",
|
|
860
|
+
"trustedNegativeAccuracy",
|
|
861
|
+
"latencyMs",
|
|
862
|
+
"calls",
|
|
863
|
+
"inputTokens",
|
|
864
|
+
"outputTokens",
|
|
865
|
+
"reasoningTokens",
|
|
866
|
+
"cachedTokens",
|
|
867
|
+
"cacheWriteTokens",
|
|
868
|
+
"costUsd"
|
|
869
|
+
]),
|
|
870
|
+
direction: z.enum(["higher", "lower"]),
|
|
871
|
+
pairedCases: nonNegativeInteger,
|
|
872
|
+
pairedClusters: nonNegativeInteger,
|
|
873
|
+
eligibleObservations: nonNegativeInteger,
|
|
874
|
+
pairedObservations: nonNegativeInteger,
|
|
875
|
+
baselineMissingObservations: nonNegativeInteger,
|
|
876
|
+
candidateMissingObservations: nonNegativeInteger,
|
|
877
|
+
asymmetricMissingObservations: nonNegativeInteger,
|
|
878
|
+
survivorOnly: z.boolean(),
|
|
879
|
+
baselineMean: nullableFiniteNumber,
|
|
880
|
+
candidateMean: nullableFiniteNumber,
|
|
881
|
+
meanDelta: nullableFiniteNumber,
|
|
882
|
+
intervalLow: nullableFiniteNumber,
|
|
883
|
+
intervalHigh: nullableFiniteNumber,
|
|
884
|
+
confidence: z.number().gt(0).lt(1),
|
|
885
|
+
resamples: positiveInteger$1,
|
|
886
|
+
minimumSampleMet: z.boolean(),
|
|
887
|
+
populationInferenceEligible: z.boolean(),
|
|
888
|
+
inferenceLimitations: stringArray
|
|
889
|
+
});
|
|
890
|
+
const comparisonSchema = z.strictObject({
|
|
891
|
+
baselineRunnerId: nonEmptyString,
|
|
892
|
+
candidateRunnerId: nonEmptyString,
|
|
893
|
+
metrics: z.array(comparisonMetricSchema)
|
|
894
|
+
});
|
|
895
|
+
const valueDistributionSchema = z.strictObject({
|
|
896
|
+
total: nonNegativeInteger,
|
|
897
|
+
missing: nonNegativeInteger,
|
|
898
|
+
counts: z.record(z.string(), nonNegativeInteger)
|
|
899
|
+
});
|
|
900
|
+
const distributionsSchema = z.strictObject({
|
|
901
|
+
class: valueDistributionSchema,
|
|
902
|
+
agent: valueDistributionSchema,
|
|
903
|
+
model: valueDistributionSchema,
|
|
904
|
+
difficulty: valueDistributionSchema,
|
|
905
|
+
solved: valueDistributionSchema
|
|
906
|
+
});
|
|
907
|
+
const selectionReportSchema = z.strictObject({
|
|
908
|
+
method: z.enum(["census", "deterministic-hash"]),
|
|
909
|
+
seed: safeInteger$1,
|
|
910
|
+
sourceCount: positiveInteger$1,
|
|
911
|
+
selectedCount: positiveInteger$1,
|
|
912
|
+
stratified: z.literal(false),
|
|
913
|
+
representativeOfInput: z.boolean(),
|
|
914
|
+
source: distributionsSchema,
|
|
915
|
+
selected: distributionsSchema
|
|
916
|
+
});
|
|
917
|
+
const verificationOutcomeSchema = z.strictObject({
|
|
918
|
+
status: z.enum([
|
|
919
|
+
"passed",
|
|
920
|
+
"failed",
|
|
921
|
+
"unavailable"
|
|
922
|
+
]),
|
|
923
|
+
reason: z.enum([
|
|
924
|
+
"missing-result",
|
|
925
|
+
"result-output-unavailable",
|
|
926
|
+
"result-parse-error",
|
|
927
|
+
"result-label-disagreement"
|
|
928
|
+
]).optional(),
|
|
929
|
+
parseError: errorSchema.optional(),
|
|
930
|
+
sources: z.array(z.strictObject({
|
|
931
|
+
path: nonEmptyString,
|
|
932
|
+
format: z.enum([
|
|
933
|
+
"terminal-bench",
|
|
934
|
+
"swe-bench",
|
|
935
|
+
"swe-multi"
|
|
936
|
+
]),
|
|
937
|
+
status: z.enum([
|
|
938
|
+
"passed",
|
|
939
|
+
"failed",
|
|
940
|
+
"unavailable"
|
|
941
|
+
])
|
|
942
|
+
})),
|
|
943
|
+
passedCheckCount: nonNegativeInteger,
|
|
944
|
+
failedCheckCount: nonNegativeInteger,
|
|
945
|
+
passedChecks: stringArray,
|
|
946
|
+
failedChecks: stringArray
|
|
947
|
+
});
|
|
948
|
+
const verificationArtifactRole = z.enum([
|
|
949
|
+
"final-test-output",
|
|
950
|
+
"final-result",
|
|
951
|
+
"final-metrics"
|
|
952
|
+
]);
|
|
953
|
+
const verificationArtifactSchema = z.strictObject({
|
|
954
|
+
traceId: nonEmptyString,
|
|
955
|
+
status: z.enum(["present", "missing"]),
|
|
956
|
+
outcome: verificationOutcomeSchema,
|
|
957
|
+
outcomeSpanId: nonEmptyString,
|
|
958
|
+
caseDirectory: nonEmptyString,
|
|
959
|
+
caseDirectoriesSearched: nonEmptyStringArray,
|
|
960
|
+
totalBytes: nonNegativeInteger,
|
|
961
|
+
maxBytes: positiveInteger$1,
|
|
962
|
+
files: z.array(z.strictObject({
|
|
963
|
+
role: verificationArtifactRole,
|
|
964
|
+
path: nonEmptyString,
|
|
965
|
+
relativePath: nonEmptyString,
|
|
966
|
+
sha256,
|
|
967
|
+
bytes: nonNegativeInteger,
|
|
968
|
+
spanId: nonEmptyString
|
|
969
|
+
})),
|
|
970
|
+
missingRoles: z.array(verificationArtifactRole),
|
|
971
|
+
searched: z.strictObject({
|
|
972
|
+
"final-test-output": stringArray,
|
|
973
|
+
"final-result": stringArray,
|
|
974
|
+
"final-metrics": stringArray
|
|
975
|
+
})
|
|
976
|
+
});
|
|
977
|
+
const verificationAvailabilitySchema = z.strictObject({
|
|
978
|
+
cases: nonNegativeInteger,
|
|
979
|
+
resultFilesPresent: nonNegativeInteger,
|
|
980
|
+
resultFilesMissing: nonNegativeInteger,
|
|
981
|
+
outcomes: z.strictObject({
|
|
982
|
+
passed: nonNegativeInteger,
|
|
983
|
+
failed: nonNegativeInteger,
|
|
984
|
+
unavailable: nonNegativeInteger
|
|
985
|
+
})
|
|
986
|
+
});
|
|
987
|
+
const codeTraceCalibrationSchema = z.strictObject({
|
|
988
|
+
protocol: z.literal("labeled-positive-and-solved-negative"),
|
|
989
|
+
rationale: nonEmptyString,
|
|
990
|
+
runners: z.array(z.strictObject({
|
|
991
|
+
runnerId: nonEmptyString,
|
|
992
|
+
selectedRuns: nonNegativeInteger,
|
|
993
|
+
positiveRuns: nonNegativeInteger,
|
|
994
|
+
trustedNegativeRuns: nonNegativeInteger,
|
|
995
|
+
unlabeledRuns: nonNegativeInteger,
|
|
996
|
+
failedLabelEmptyRuns: nonNegativeInteger,
|
|
997
|
+
unknownLabelEmptyRuns: nonNegativeInteger,
|
|
998
|
+
completedRuns: nonNegativeInteger,
|
|
999
|
+
failedRuns: nonNegativeInteger,
|
|
1000
|
+
expectedIncorrectSteps: nonNegativeInteger,
|
|
1001
|
+
predictedIncorrectSteps: nonNegativeInteger,
|
|
1002
|
+
matchedIncorrectSteps: nonNegativeInteger,
|
|
1003
|
+
officialAllRowF1: nullableRate,
|
|
1004
|
+
officialAllRowRuns: nonNegativeInteger,
|
|
1005
|
+
precision: nullableRate,
|
|
1006
|
+
recall: nullableRate,
|
|
1007
|
+
f1: nullableRate,
|
|
1008
|
+
trustedNegativeFalsePositiveRate: nullableRate,
|
|
1009
|
+
trustedNegativeFailureRate: nullableRate,
|
|
1010
|
+
unlabeledPredictionRate: nullableRate,
|
|
1011
|
+
unlabeledFailureRate: nullableRate
|
|
1012
|
+
}))
|
|
1013
|
+
});
|
|
1014
|
+
const agentRxCalibrationSchema = z.strictObject({
|
|
1015
|
+
protocol: z.literal("official-agentrx-root-cause"),
|
|
1016
|
+
upstreamRevision: revision,
|
|
1017
|
+
rationale: nonEmptyString,
|
|
1018
|
+
runners: z.array(z.strictObject({
|
|
1019
|
+
runnerId: nonEmptyString,
|
|
1020
|
+
selectedRuns: nonNegativeInteger,
|
|
1021
|
+
completedRuns: nonNegativeInteger,
|
|
1022
|
+
failedRuns: nonNegativeInteger,
|
|
1023
|
+
predictedRuns: nonNegativeInteger,
|
|
1024
|
+
missingPredictionRuns: nonNegativeInteger,
|
|
1025
|
+
exactStepAccuracy: nullableRate,
|
|
1026
|
+
stepAccuracyWithin1: nullableRate,
|
|
1027
|
+
stepAccuracyWithin2: nullableRate,
|
|
1028
|
+
stepAccuracyWithin3: nullableRate,
|
|
1029
|
+
stepAccuracyWithin4: nullableRate,
|
|
1030
|
+
stepAccuracyWithin5: nullableRate,
|
|
1031
|
+
meanStepDistance: nonNegativeNumber.nullable(),
|
|
1032
|
+
normalizedMeanStepDistance: nullableRate,
|
|
1033
|
+
normalizedDistanceRuns: nonNegativeInteger,
|
|
1034
|
+
normalizedDistanceUnknownRuns: nonNegativeInteger,
|
|
1035
|
+
rootCauseCategoryAccuracy: nullableRate,
|
|
1036
|
+
anyFailureCategoryAccuracy: nullableRate,
|
|
1037
|
+
earliestFailureCategoryAccuracy: nullableRate,
|
|
1038
|
+
terminalFailureCategoryAccuracy: nullableRate
|
|
1039
|
+
}))
|
|
1040
|
+
});
|
|
1041
|
+
const artifactSchema = z.strictObject({
|
|
1042
|
+
kind: z.literal("agent-eval/analyst-benchmark-result"),
|
|
1043
|
+
runIdentitySha256: sha256,
|
|
1044
|
+
inputs: z.strictObject({
|
|
1045
|
+
dataset: z.enum(["agentrx", "codetracebench"]),
|
|
1046
|
+
datasetRevision: revision,
|
|
1047
|
+
datasetSplit: nonEmptyString,
|
|
1048
|
+
labelsSha256: sha256,
|
|
1049
|
+
sourceRowCount: positiveInteger$1,
|
|
1050
|
+
traceFiles: z.array(z.strictObject({
|
|
1051
|
+
traceId: nonEmptyString,
|
|
1052
|
+
relativePath: nonEmptyString,
|
|
1053
|
+
sha256
|
|
1054
|
+
})),
|
|
1055
|
+
verificationArtifacts: z.array(verificationArtifactSchema),
|
|
1056
|
+
verificationAvailability: verificationAvailabilitySchema,
|
|
1057
|
+
selection: z.strictObject({
|
|
1058
|
+
limit: positiveInteger$1,
|
|
1059
|
+
seed: safeInteger$1,
|
|
1060
|
+
selectedCaseIds: nonEmptyStringArray.min(1),
|
|
1061
|
+
report: selectionReportSchema
|
|
1062
|
+
}),
|
|
1063
|
+
execution: z.strictObject({
|
|
1064
|
+
repetitions: positiveInteger$1,
|
|
1065
|
+
concurrency: positiveInteger$1,
|
|
1066
|
+
model: nonEmptyString,
|
|
1067
|
+
maxOutputTokens: positiveInteger$1,
|
|
1068
|
+
timeoutMs: positiveInteger$1,
|
|
1069
|
+
maxCostUsd: nonNegativeNumber,
|
|
1070
|
+
maxArtifactBytes: positiveInteger$1,
|
|
1071
|
+
analystProtocolSha256: sha256,
|
|
1072
|
+
implementationSha256: sha256,
|
|
1073
|
+
dependencyLockSha256: sha256
|
|
1074
|
+
})
|
|
1075
|
+
}),
|
|
1076
|
+
result: resultSchema,
|
|
1077
|
+
comparisons: z.array(comparisonSchema),
|
|
1078
|
+
codeTraceCalibration: codeTraceCalibrationSchema.optional(),
|
|
1079
|
+
agentRxCalibration: agentRxCalibrationSchema.optional()
|
|
1080
|
+
}).superRefine((artifact, context) => {
|
|
1081
|
+
const isCodeTrace = artifact.inputs.dataset === "codetracebench";
|
|
1082
|
+
if (isCodeTrace && !artifact.codeTraceCalibration) context.addIssue({
|
|
1083
|
+
code: "custom",
|
|
1084
|
+
path: ["codeTraceCalibration"],
|
|
1085
|
+
message: "is required for CodeTraceBench artifacts"
|
|
1086
|
+
});
|
|
1087
|
+
if (isCodeTrace && artifact.agentRxCalibration) context.addIssue({
|
|
1088
|
+
code: "custom",
|
|
1089
|
+
path: ["agentRxCalibration"],
|
|
1090
|
+
message: "is not allowed for CodeTraceBench artifacts"
|
|
1091
|
+
});
|
|
1092
|
+
if (!isCodeTrace && !artifact.agentRxCalibration) context.addIssue({
|
|
1093
|
+
code: "custom",
|
|
1094
|
+
path: ["agentRxCalibration"],
|
|
1095
|
+
message: "is required for AgentRx artifacts"
|
|
1096
|
+
});
|
|
1097
|
+
if (!isCodeTrace && artifact.codeTraceCalibration) context.addIssue({
|
|
1098
|
+
code: "custom",
|
|
1099
|
+
path: ["codeTraceCalibration"],
|
|
1100
|
+
message: "is not allowed for AgentRx artifacts"
|
|
1101
|
+
});
|
|
1102
|
+
});
|
|
1103
|
+
function assertAnalystBenchmarkObservation(value, context) {
|
|
1104
|
+
assertSchema(observationSchema, value, context);
|
|
1105
|
+
}
|
|
1106
|
+
function assertAnalystBenchmarkArtifact(value, context) {
|
|
1107
|
+
assertSchema(artifactSchema, value, context);
|
|
1108
|
+
}
|
|
1109
|
+
function assertSchema(schema, value, context) {
|
|
1110
|
+
const result = schema.safeParse(value);
|
|
1111
|
+
if (result.success) return;
|
|
1112
|
+
throw new TypeError(formatIssue(result.error.issues[0], context));
|
|
1113
|
+
}
|
|
1114
|
+
function formatIssue(issue, context) {
|
|
1115
|
+
const path = issue.path.length === 0 ? context : `${context}.${issue.path.join(".")}`;
|
|
1116
|
+
if (issue.code === "unrecognized_keys") return `${path} contains unknown field '${issue.keys[0]}'`;
|
|
1117
|
+
return `${path} ${issue.message}`;
|
|
1118
|
+
}
|
|
1119
|
+
//#endregion
|
|
1120
|
+
//#region src/analyst/benchmark-command-artifact.ts
|
|
1121
|
+
const ANALYST_BENCHMARK_MANIFEST_FILE = "manifest.json";
|
|
1122
|
+
const ANALYST_BENCHMARK_OBSERVATIONS_FILE = "observations.jsonl";
|
|
1123
|
+
const ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE = "run.local.json";
|
|
1124
|
+
const ANALYST_BENCHMARK_COST_LEDGER_FILE = "cost-ledger.jsonl";
|
|
1125
|
+
function observationKey(observation) {
|
|
1126
|
+
return `${observation.runnerId}\u0000${observation.caseId}\u0000${observation.repetition}`;
|
|
1127
|
+
}
|
|
1128
|
+
function parseJson(text, source) {
|
|
1129
|
+
try {
|
|
1130
|
+
return JSON.parse(text);
|
|
1131
|
+
} catch {
|
|
1132
|
+
throw new Error(`invalid JSON in ${source}`);
|
|
1133
|
+
}
|
|
1134
|
+
}
|
|
1135
|
+
function assertExactKeys(value, allowed, context, optional = []) {
|
|
1136
|
+
const allowedSet = new Set(allowed);
|
|
1137
|
+
const optionalSet = new Set(optional);
|
|
1138
|
+
for (const key of Object.keys(value)) if (!allowedSet.has(key)) throw new TypeError(`${context} contains unknown field '${key}'`);
|
|
1139
|
+
for (const key of allowed) if (!optionalSet.has(key) && !(key in value)) throw new TypeError(`${context} is missing field '${key}'`);
|
|
1140
|
+
}
|
|
1141
|
+
function digestCanonical(value) {
|
|
1142
|
+
return createHash("sha256").update(canonicalJson(value)).digest("hex");
|
|
1143
|
+
}
|
|
1144
|
+
function canonicalJson(value) {
|
|
1145
|
+
if (value === null || typeof value === "string" || typeof value === "boolean") return JSON.stringify(value);
|
|
1146
|
+
if (typeof value === "number") {
|
|
1147
|
+
if (!Number.isFinite(value)) throw new TypeError("cannot hash a non-finite number");
|
|
1148
|
+
return JSON.stringify(value);
|
|
1149
|
+
}
|
|
1150
|
+
if (Array.isArray(value)) return `[${value.map(canonicalJson).join(",")}]`;
|
|
1151
|
+
if (isRecord$1(value)) return `{${Object.keys(value).filter((key) => value[key] !== void 0).sort().map((key) => `${JSON.stringify(key)}:${canonicalJson(value[key])}`).join(",")}}`;
|
|
1152
|
+
throw new TypeError(`cannot hash ${typeof value}`);
|
|
1153
|
+
}
|
|
1154
|
+
function isRecord$1(value) {
|
|
1155
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
1156
|
+
}
|
|
1157
|
+
function isSha256(value) {
|
|
1158
|
+
return typeof value === "string" && /^[a-f0-9]{64}$/.test(value);
|
|
1159
|
+
}
|
|
1160
|
+
//#endregion
|
|
1161
|
+
//#region src/analyst/benchmark-implementation.ts
|
|
1162
|
+
const ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM = "sha256-canonical-source-manifest";
|
|
1163
|
+
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM = "sha256-canonical-file-manifest";
|
|
1164
|
+
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze(["package.json", "pnpm-lock.yaml"]);
|
|
1165
|
+
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "d06e3e3f171cc6bd5cf36b14325f507f34b1c6a9d5129ce8a85bf3c1681f8223";
|
|
1166
|
+
/** The published benchmark evidence was produced at this package version.
|
|
1167
|
+
* A release changes package.json's version field, which is part of the lock
|
|
1168
|
+
* manifest but cannot change benchmark behavior, so the evidence stays bound
|
|
1169
|
+
* to the digest at its creation. A test proves the current lock differs from
|
|
1170
|
+
* the evidence lock by the version stamp alone; any real dependency change
|
|
1171
|
+
* still forces a new benchmark run or explicit retirement of the evidence. */
|
|
1172
|
+
const ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION = "0.137.0";
|
|
1173
|
+
const ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 = "1e03f2daed356d60316aabefb407ec1e437ac94d408d61eea4ae096e9c6fbb5b";
|
|
1174
|
+
const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
1175
|
+
"src/analyst/benchmark-agentrx-calibration.ts",
|
|
1176
|
+
"src/analyst/benchmark-command-artifact.ts",
|
|
1177
|
+
"src/analyst/benchmark-command-persistence.ts",
|
|
1178
|
+
"src/analyst/benchmark-command-result.ts",
|
|
1179
|
+
"src/analyst/benchmark-command-validation.ts",
|
|
1180
|
+
"src/analyst/benchmark-command.ts",
|
|
1181
|
+
"src/analyst/benchmark-comparison.ts",
|
|
1182
|
+
"src/analyst/benchmark-dataset-agentrx.ts",
|
|
1183
|
+
"src/analyst/benchmark-dataset-codetrace.ts",
|
|
1184
|
+
"src/analyst/benchmark-dataset-utils.ts",
|
|
1185
|
+
"src/analyst/benchmark-datasets.ts",
|
|
1186
|
+
"src/analyst/benchmark-evidence-validation.ts",
|
|
1187
|
+
"src/analyst/benchmark-public-adapters.ts",
|
|
1188
|
+
"src/analyst/benchmark-public-calibration.ts",
|
|
1189
|
+
"src/analyst/benchmark-public-data.ts",
|
|
1190
|
+
"src/analyst/benchmark-public-errors.ts",
|
|
1191
|
+
"src/analyst/benchmark-public-model.ts",
|
|
1192
|
+
"src/analyst/benchmark-public-types.ts",
|
|
1193
|
+
"src/analyst/benchmark-real-model.ts",
|
|
1194
|
+
"src/analyst/benchmark-report.ts",
|
|
1195
|
+
"src/analyst/benchmark-response-cache.ts",
|
|
1196
|
+
"src/analyst/benchmark-scoring.ts",
|
|
1197
|
+
"src/analyst/benchmark-summary.ts",
|
|
1198
|
+
"src/analyst/benchmark-verification-artifacts.ts",
|
|
1199
|
+
"src/analyst/benchmark-verification-outcome.ts",
|
|
1200
|
+
"src/analyst/benchmark.ts",
|
|
1201
|
+
"src/analyst/types.ts",
|
|
1202
|
+
"src/analyst/usage-receipt.ts",
|
|
1203
|
+
"src/campaign/search-ledger-errors.ts",
|
|
1204
|
+
"src/campaign/search-ledger-file.ts",
|
|
1205
|
+
"src/campaign/single-run-lock.ts",
|
|
1206
|
+
"src/campaign/storage.ts",
|
|
1207
|
+
"src/concurrency.ts",
|
|
1208
|
+
"src/cost-ledger.ts",
|
|
1209
|
+
"src/errors.ts",
|
|
1210
|
+
"src/judge-calibration.ts",
|
|
1211
|
+
"src/ledger-core/atomic-file-lock.ts",
|
|
1212
|
+
"src/ledger-core/canonical.ts",
|
|
1213
|
+
"src/ledger-core/index.ts",
|
|
1214
|
+
"src/ledger-core/journal-file.ts",
|
|
1215
|
+
"src/ledger-core/journal.ts",
|
|
1216
|
+
"src/ledger-core/trusted-head.ts",
|
|
1217
|
+
"src/llm-client.ts",
|
|
1218
|
+
"src/math/normal.ts",
|
|
1219
|
+
"src/math/special-functions.ts",
|
|
1220
|
+
"src/math/student-t.ts",
|
|
1221
|
+
"src/metrics.ts",
|
|
1222
|
+
"src/statistics.ts",
|
|
1223
|
+
"src/trace-analyst/errors.ts",
|
|
1224
|
+
"src/trace-analyst/otlp-span.ts",
|
|
1225
|
+
"src/trace-analyst/shared-abortable-task.ts",
|
|
1226
|
+
"src/trace-analyst/store-boundary.ts",
|
|
1227
|
+
"src/trace-analyst/store-bounds.ts",
|
|
1228
|
+
"src/trace-analyst/store-contract.ts",
|
|
1229
|
+
"src/trace-analyst/store-otlp.ts",
|
|
1230
|
+
"src/trace-analyst/store-schemas.ts",
|
|
1231
|
+
"src/trace-analyst/store.ts",
|
|
1232
|
+
"src/trace-analyst/types.ts",
|
|
1233
|
+
"src/trace/attribute-vocabulary.ts",
|
|
1234
|
+
"src/trace/otlp-attributes.ts",
|
|
1235
|
+
"src/trace/raw-provider-sink.ts"
|
|
1236
|
+
]);
|
|
1237
|
+
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "4dba263b6256a30d56c7fdb2d992d3a953c0035d731f359b704db806f68f75ac";
|
|
1238
|
+
function analystBenchmarkImplementationDigest() {
|
|
1239
|
+
return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256;
|
|
1240
|
+
}
|
|
1241
|
+
function analystBenchmarkDependencyLockDigest() {
|
|
1242
|
+
return ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256;
|
|
1243
|
+
}
|
|
1244
|
+
//#endregion
|
|
1245
|
+
//#region src/analyst/benchmark-evidence-validation.ts
|
|
1246
|
+
const MIN_ACTION_EXCERPT_CHARACTERS = 12;
|
|
1247
|
+
const MAX_LABEL_SCAN_DEPTH = 64;
|
|
1248
|
+
const MAX_SERIALIZED_JSON_SCAN_BYTES = 64 * 1024;
|
|
1249
|
+
const MAX_SERIALIZED_JSON_SCAN_DEPTH = 8;
|
|
1250
|
+
const BENCHMARK_LABEL_KEYS = /* @__PURE__ */ new Set([
|
|
1251
|
+
"category_reason",
|
|
1252
|
+
"failure_category",
|
|
1253
|
+
"failure_summary",
|
|
1254
|
+
"incorrect_stages",
|
|
1255
|
+
"incorrect_step_ids",
|
|
1256
|
+
"root_cause",
|
|
1257
|
+
"root_cause_failure_id",
|
|
1258
|
+
"root_cause_reason",
|
|
1259
|
+
"step_reason",
|
|
1260
|
+
"unuseful_step_ids"
|
|
1261
|
+
]);
|
|
1262
|
+
const BENCHMARK_LABEL_KEY_TOKENS = [...BENCHMARK_LABEL_KEYS].sort((left, right) => right.length - left.length);
|
|
1263
|
+
const BENCHMARK_LABEL_PATH_MARKERS = [
|
|
1264
|
+
"bench_manifest.verified",
|
|
1265
|
+
"codetracer_labels.json",
|
|
1266
|
+
"/ground_truth/",
|
|
1267
|
+
"\\ground_truth\\"
|
|
1268
|
+
];
|
|
1269
|
+
function assertNoBenchmarkLabelsInTrace(options) {
|
|
1270
|
+
let scannedValues = 0;
|
|
1271
|
+
for (const [index, line] of options.otlpText.split(/\r?\n/).entries()) {
|
|
1272
|
+
if (!line.trim()) continue;
|
|
1273
|
+
let value;
|
|
1274
|
+
try {
|
|
1275
|
+
value = JSON.parse(line);
|
|
1276
|
+
} catch (error) {
|
|
1277
|
+
throw new TypeError(`trace '${options.traceId}' line ${index + 1} is invalid JSON: ${error instanceof Error ? error.message : String(error)}`);
|
|
1278
|
+
}
|
|
1279
|
+
scannedValues += scanValue(value, options.traceId, `$line[${index + 1}]`, 0, 0);
|
|
1280
|
+
}
|
|
1281
|
+
if (scannedValues === 0) throw new Error(`trace '${options.traceId}' contains no JSON values`);
|
|
1282
|
+
return {
|
|
1283
|
+
passed: true,
|
|
1284
|
+
scannedBytes: Buffer.byteLength(options.otlpText),
|
|
1285
|
+
scannedValues
|
|
1286
|
+
};
|
|
1287
|
+
}
|
|
1288
|
+
function assertNoBenchmarkLabelsInArtifact(options) {
|
|
1289
|
+
const normalizedPath = options.relativePath.toLowerCase();
|
|
1290
|
+
for (const marker of BENCHMARK_LABEL_PATH_MARKERS) if (normalizedPath.includes(marker.toLowerCase())) throw new Error(`trace '${options.traceId}' verification artifact path contains benchmark label marker '${marker}'`);
|
|
1291
|
+
const normalizedContent = options.content.toLowerCase();
|
|
1292
|
+
for (const key of BENCHMARK_LABEL_KEYS) if (normalizedContent.includes(key)) throw new Error(`trace '${options.traceId}' verification artifact contains benchmark label key '${key}'`);
|
|
1293
|
+
for (const marker of BENCHMARK_LABEL_PATH_MARKERS) if (normalizedContent.includes(marker.toLowerCase())) throw new Error(`trace '${options.traceId}' verification artifact contains benchmark label path marker '${marker}'`);
|
|
1294
|
+
}
|
|
1295
|
+
async function validateCodeTraceFindingEvidence(options) {
|
|
1296
|
+
const citations = options.findings.flatMap((finding) => finding.evidence_refs.map((evidence) => ({
|
|
1297
|
+
evidence,
|
|
1298
|
+
findingId: finding.finding_id,
|
|
1299
|
+
location: codeTraceStepFromEvidence(evidence.uri)
|
|
1300
|
+
})));
|
|
1301
|
+
if (citations.length === 0) return;
|
|
1302
|
+
for (const citation of citations) if (!citation.location || citation.location.traceId !== options.trajectoryId) throw new Error(`model finding '${citation.findingId}' cites non-case evidence '${citation.evidence.uri}'`);
|
|
1303
|
+
const spanIds = [...new Set(citations.map((citation) => `step-${citation.location.step}`))];
|
|
1304
|
+
const spans = /* @__PURE__ */ new Map();
|
|
1305
|
+
for (let offset = 0; offset < spanIds.length; offset += TRACE_ANALYSIS_LIMITS.viewSpans) {
|
|
1306
|
+
const requested = spanIds.slice(offset, offset + TRACE_ANALYSIS_LIMITS.viewSpans);
|
|
1307
|
+
const result = await options.store.viewSpans({
|
|
1308
|
+
trace_id: options.trajectoryId,
|
|
1309
|
+
span_ids: requested
|
|
1310
|
+
}, options.signal ? { signal: options.signal } : void 0);
|
|
1311
|
+
if (result.missing_span_ids.length > 0 || result.omitted_span_ids.length > 0) {
|
|
1312
|
+
const unavailable = [...result.missing_span_ids, ...result.omitted_span_ids];
|
|
1313
|
+
throw new Error(`model finding evidence is unavailable in the case trace: ${unavailable.join(", ")}`);
|
|
1314
|
+
}
|
|
1315
|
+
for (const span of result.spans) spans.set(span.span_id, span);
|
|
1316
|
+
}
|
|
1317
|
+
for (const citation of citations) {
|
|
1318
|
+
const spanId = `step-${citation.location.step}`;
|
|
1319
|
+
const span = spans.get(spanId);
|
|
1320
|
+
if (!span) throw new Error(`model finding '${citation.findingId}' cites missing span '${spanId}'`);
|
|
1321
|
+
if (span.kind !== "LLM") throw new Error(`model finding '${citation.findingId}' cites '${spanId}', which is ${span.kind}, not an assistant LLM span`);
|
|
1322
|
+
assertExactActionExcerpt(citation.findingId, citation.evidence, spanId, span.attributes.content);
|
|
1323
|
+
}
|
|
1324
|
+
}
|
|
1325
|
+
async function resolveAssistantStepEvidence(options) {
|
|
1326
|
+
const steps = [...new Set(options.steps)];
|
|
1327
|
+
for (const step of steps) if (!Number.isSafeInteger(step) || step < 1) throw new TypeError(`assistant evidence step must be a positive safe integer: ${step}`);
|
|
1328
|
+
const spanIds = steps.map((step) => `step-${step}`);
|
|
1329
|
+
const spans = /* @__PURE__ */ new Map();
|
|
1330
|
+
for (let offset = 0; offset < spanIds.length; offset += TRACE_ANALYSIS_LIMITS.viewSpans) {
|
|
1331
|
+
const requested = spanIds.slice(offset, offset + TRACE_ANALYSIS_LIMITS.viewSpans);
|
|
1332
|
+
const result = await options.store.viewSpans({
|
|
1333
|
+
trace_id: options.trajectoryId,
|
|
1334
|
+
span_ids: requested
|
|
1335
|
+
}, options.signal ? { signal: options.signal } : void 0);
|
|
1336
|
+
if (result.missing_span_ids.length > 0 || result.omitted_span_ids.length > 0) {
|
|
1337
|
+
const unavailable = [...result.missing_span_ids, ...result.omitted_span_ids];
|
|
1338
|
+
throw new Error(`model selected unavailable assistant steps: ${unavailable.join(", ")}`);
|
|
1339
|
+
}
|
|
1340
|
+
for (const span of result.spans) spans.set(span.span_id, span);
|
|
1341
|
+
}
|
|
1342
|
+
const evidence = /* @__PURE__ */ new Map();
|
|
1343
|
+
for (const step of steps) {
|
|
1344
|
+
const spanId = `step-${step}`;
|
|
1345
|
+
const span = spans.get(spanId);
|
|
1346
|
+
if (!span) throw new Error(`model selected missing assistant step '${spanId}'`);
|
|
1347
|
+
if (span.kind !== "LLM") throw new Error(`model selected '${spanId}', which is ${span.kind}, not an assistant LLM span`);
|
|
1348
|
+
const content = span.attributes.content;
|
|
1349
|
+
if (typeof content !== "string" || content.trim().length === 0) throw new Error(`model selected '${spanId}' without action content`);
|
|
1350
|
+
evidence.set(step, {
|
|
1351
|
+
kind: "span",
|
|
1352
|
+
uri: codeTraceStepEvidenceUri(options.trajectoryId, step),
|
|
1353
|
+
excerpt: content.trim().slice(0, 512)
|
|
1354
|
+
});
|
|
1355
|
+
}
|
|
1356
|
+
return evidence;
|
|
1357
|
+
}
|
|
1358
|
+
function scanValue(value, traceId, path, depth, serializedDepth) {
|
|
1359
|
+
if (depth > MAX_LABEL_SCAN_DEPTH) throw new Error(`trace '${traceId}' exceeds benchmark label scan depth at ${path}`);
|
|
1360
|
+
if (Array.isArray(value)) return 1 + value.reduce((count, entry, index) => count + scanValue(entry, traceId, `${path}[${index}]`, depth + 1, serializedDepth), 0);
|
|
1361
|
+
if (typeof value === "object" && value !== null) {
|
|
1362
|
+
let count = 1;
|
|
1363
|
+
for (const [key, entry] of Object.entries(value)) {
|
|
1364
|
+
if (BENCHMARK_LABEL_KEYS.has(key.toLowerCase())) throw new Error(`trace '${traceId}' exposes benchmark label key '${key}' at ${path}`);
|
|
1365
|
+
count += scanValue(entry, traceId, `${path}.${key}`, depth + 1, serializedDepth);
|
|
1366
|
+
}
|
|
1367
|
+
return count;
|
|
1368
|
+
}
|
|
1369
|
+
if (typeof value === "string") {
|
|
1370
|
+
const normalized = value.toLowerCase();
|
|
1371
|
+
const labelKey = BENCHMARK_LABEL_KEY_TOKENS.find((candidate) => normalized.includes(candidate));
|
|
1372
|
+
if (labelKey) throw new Error(`trace '${traceId}' exposes benchmark label key '${labelKey}' inside a string at ${path}`);
|
|
1373
|
+
const marker = BENCHMARK_LABEL_PATH_MARKERS.find((candidate) => normalized.includes(candidate));
|
|
1374
|
+
if (marker) throw new Error(`trace '${traceId}' exposes benchmark label path marker '${marker}' at ${path}`);
|
|
1375
|
+
const trimmed = value.trim();
|
|
1376
|
+
if (serializedDepth < MAX_SERIALIZED_JSON_SCAN_DEPTH && Buffer.byteLength(trimmed) <= MAX_SERIALIZED_JSON_SCAN_BYTES && looksLikeSerializedJson(trimmed)) {
|
|
1377
|
+
let parsed;
|
|
1378
|
+
try {
|
|
1379
|
+
parsed = JSON.parse(trimmed);
|
|
1380
|
+
} catch {
|
|
1381
|
+
return 1;
|
|
1382
|
+
}
|
|
1383
|
+
return 1 + scanValue(parsed, traceId, `${path}<serialized-json>`, depth + 1, serializedDepth + 1);
|
|
1384
|
+
}
|
|
1385
|
+
}
|
|
1386
|
+
return 1;
|
|
1387
|
+
}
|
|
1388
|
+
function looksLikeSerializedJson(value) {
|
|
1389
|
+
return value.startsWith("{") && value.endsWith("}") || value.startsWith("[") && value.endsWith("]") || value.startsWith("\"") && value.endsWith("\"");
|
|
1390
|
+
}
|
|
1391
|
+
function codeTraceStepFromEvidence(uri) {
|
|
1392
|
+
const match = /^trace:\/\/([^/]+)\/span\/step-(\d+)$/.exec(uri);
|
|
1393
|
+
if (!match) return null;
|
|
1394
|
+
try {
|
|
1395
|
+
const traceId = decodeURIComponent(match[1]);
|
|
1396
|
+
const step = Number(match[2]);
|
|
1397
|
+
return traceId && Number.isSafeInteger(step) && step > 0 ? {
|
|
1398
|
+
traceId,
|
|
1399
|
+
step
|
|
1400
|
+
} : null;
|
|
1401
|
+
} catch {
|
|
1402
|
+
return null;
|
|
1403
|
+
}
|
|
1404
|
+
}
|
|
1405
|
+
function codeTraceStepEvidenceUri(traceId, step) {
|
|
1406
|
+
return `trace://${encodeURIComponent(traceId)}/span/step-${step}`;
|
|
1407
|
+
}
|
|
1408
|
+
function assertExactActionExcerpt(findingId, evidence, spanId, content) {
|
|
1409
|
+
if (typeof content !== "string" || content.length === 0) throw new Error(`model finding '${findingId}' cites '${spanId}' without action content`);
|
|
1410
|
+
const excerpt = evidence.excerpt?.trim();
|
|
1411
|
+
if (!excerpt) throw new Error(`model finding '${findingId}' must quote action content from '${spanId}'`);
|
|
1412
|
+
const requiredLength = Math.min(MIN_ACTION_EXCERPT_CHARACTERS, content.trim().length);
|
|
1413
|
+
if (excerpt.length < requiredLength) throw new Error(`model finding '${findingId}' excerpt for '${spanId}' is too short; expected at least ${requiredLength} characters`);
|
|
1414
|
+
if (!content.includes(excerpt)) throw new Error(`model finding '${findingId}' excerpt is not present in '${spanId}' action content`);
|
|
1415
|
+
}
|
|
1416
|
+
//#endregion
|
|
1417
|
+
//#region src/analyst/benchmark-public-adapters.ts
|
|
1418
|
+
function emptyPublicBenchmarkRunner() {
|
|
1419
|
+
return {
|
|
1420
|
+
id: "empty",
|
|
1421
|
+
analyze() {
|
|
1422
|
+
return {
|
|
1423
|
+
findings: [],
|
|
1424
|
+
usage: {
|
|
1425
|
+
calls: 0,
|
|
1426
|
+
tokens: {
|
|
1427
|
+
input: 0,
|
|
1428
|
+
output: 0
|
|
1429
|
+
},
|
|
1430
|
+
cost: {
|
|
1431
|
+
kind: "observed",
|
|
1432
|
+
usd: 0
|
|
1433
|
+
}
|
|
1434
|
+
},
|
|
1435
|
+
metadata: { baseline: "emit-no-findings" }
|
|
1436
|
+
};
|
|
1437
|
+
}
|
|
1438
|
+
};
|
|
1439
|
+
}
|
|
1440
|
+
function adaptPublicBenchmarkFindings(dataset, trajectoryId, findings, analystId) {
|
|
1441
|
+
return dataset === "agentrx" ? adaptAgentRxFindings(trajectoryId, findings, analystId) : adaptCodeTraceFindings(trajectoryId, findings, analystId);
|
|
1442
|
+
}
|
|
1443
|
+
function adaptAgentRxFindings(trajectoryId, findings, analystId) {
|
|
1444
|
+
if (findings.length === 0) return [];
|
|
1445
|
+
if (findings.length !== 1) throw new Error(`AgentRx model analyst must emit zero or one root cause, received ${findings.length}`);
|
|
1446
|
+
const source = findings[0];
|
|
1447
|
+
if (!source.subject) throw new Error("AgentRx model analyst finding is missing its failure-category subject");
|
|
1448
|
+
const steps = exactFindingSteps(trajectoryId, source);
|
|
1449
|
+
if (steps.length !== 1) throw new Error(`AgentRx model analyst must cite exactly one root-cause step, received ${steps.length}`);
|
|
1450
|
+
const [adapted] = agentRxPredictionsToFindings(trajectoryId, [{
|
|
1451
|
+
failure_case: source.subject,
|
|
1452
|
+
step_number: steps[0],
|
|
1453
|
+
description: source.rationale ?? source.claim
|
|
1454
|
+
}], {
|
|
1455
|
+
analystId,
|
|
1456
|
+
producedAt: source.produced_at,
|
|
1457
|
+
confidence: source.confidence
|
|
1458
|
+
});
|
|
1459
|
+
if (!adapted) throw new Error("AgentRx output adapter produced no root-cause finding");
|
|
1460
|
+
return [{
|
|
1461
|
+
...adapted,
|
|
1462
|
+
metadata: {
|
|
1463
|
+
...adapted.metadata,
|
|
1464
|
+
sourceFindingId: source.finding_id
|
|
1465
|
+
}
|
|
1466
|
+
}];
|
|
1467
|
+
}
|
|
1468
|
+
function adaptCodeTraceFindings(trajectoryId, findings, analystId) {
|
|
1469
|
+
const clean = findings.filter((finding) => finding.subject === "clean");
|
|
1470
|
+
if (clean.length > 0) {
|
|
1471
|
+
if (findings.length !== 1) throw new Error("CodeTraceBench model analyst mixed a clean verdict with incorrect steps");
|
|
1472
|
+
exactFindingSteps(trajectoryId, clean[0]);
|
|
1473
|
+
return [];
|
|
1474
|
+
}
|
|
1475
|
+
const byStep = /* @__PURE__ */ new Map();
|
|
1476
|
+
for (const source of findings) {
|
|
1477
|
+
const steps = exactFindingSteps(trajectoryId, source);
|
|
1478
|
+
for (const step of steps) {
|
|
1479
|
+
if (byStep.has(step)) continue;
|
|
1480
|
+
byStep.set(step, makeFinding({
|
|
1481
|
+
analyst_id: analystId,
|
|
1482
|
+
area: "incorrect",
|
|
1483
|
+
subject: `incorrect-step-${step}`,
|
|
1484
|
+
claim: `Step ${step} is incorrect. ${source.claim}`,
|
|
1485
|
+
rationale: source.rationale,
|
|
1486
|
+
severity: source.severity,
|
|
1487
|
+
confidence: source.confidence,
|
|
1488
|
+
evidence_refs: [{
|
|
1489
|
+
kind: "span",
|
|
1490
|
+
uri: codeTraceStepEvidenceUri(trajectoryId, step),
|
|
1491
|
+
excerpt: source.evidence_refs.find((evidence) => evidence.uri === codeTraceStepEvidenceUri(trajectoryId, step))?.excerpt
|
|
1492
|
+
}],
|
|
1493
|
+
recommended_action: source.recommended_action,
|
|
1494
|
+
metadata: { sourceFindingId: source.finding_id },
|
|
1495
|
+
produced_at: source.produced_at,
|
|
1496
|
+
id_basis: `incorrect-step-${step}`
|
|
1497
|
+
}));
|
|
1498
|
+
}
|
|
1499
|
+
}
|
|
1500
|
+
return [...byStep].sort(([left], [right]) => left - right).map(([, finding]) => finding);
|
|
1501
|
+
}
|
|
1502
|
+
function exactFindingSteps(trajectoryId, finding) {
|
|
1503
|
+
if (finding.evidence_refs.length === 0) throw new Error(`model finding '${finding.finding_id}' has no step evidence`);
|
|
1504
|
+
const steps = finding.evidence_refs.map((evidence) => {
|
|
1505
|
+
const parsed = codeTraceStepFromEvidence(evidence.uri);
|
|
1506
|
+
if (!parsed || parsed.traceId !== trajectoryId) throw new Error(`model finding '${finding.finding_id}' cites non-case evidence '${evidence.uri}'`);
|
|
1507
|
+
return parsed.step;
|
|
1508
|
+
});
|
|
1509
|
+
return [...new Set(steps)];
|
|
1510
|
+
}
|
|
1511
|
+
//#endregion
|
|
1512
|
+
//#region src/analyst/benchmark-public-types.ts
|
|
1513
|
+
function requiredString(value, field) {
|
|
1514
|
+
const trimmed = value.trim();
|
|
1515
|
+
if (!trimmed) throw new TypeError(`${field} must be a non-empty string`);
|
|
1516
|
+
return trimmed;
|
|
1517
|
+
}
|
|
1518
|
+
function positiveSafeInteger(value, field) {
|
|
1519
|
+
if (!Number.isSafeInteger(value) || value < 1) throw new RangeError(`${field} must be a positive safe integer`);
|
|
1520
|
+
return value;
|
|
1521
|
+
}
|
|
1522
|
+
function safeInteger(value, field) {
|
|
1523
|
+
if (!Number.isSafeInteger(value)) throw new RangeError(`${field} must be a safe integer`);
|
|
1524
|
+
return value;
|
|
1525
|
+
}
|
|
1526
|
+
function isRecord(value) {
|
|
1527
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
1528
|
+
}
|
|
1529
|
+
//#endregion
|
|
1530
|
+
//#region src/analyst/benchmark-verification-outcome.ts
|
|
1531
|
+
const MAX_REPORTED_CHECKS = 20;
|
|
1532
|
+
const SWE_MULTI_NO_TEST_RESULTS = "After applying the fix patch, no test results were captured when executing the test command.";
|
|
1533
|
+
const checkNameSchema = z.string().min(1);
|
|
1534
|
+
const checkListSchema = z.array(checkNameSchema).superRefine((checks, context) => {
|
|
1535
|
+
const seen = /* @__PURE__ */ new Set();
|
|
1536
|
+
for (const [index, check] of checks.entries()) {
|
|
1537
|
+
if (seen.has(check)) context.addIssue({
|
|
1538
|
+
code: "custom",
|
|
1539
|
+
path: [index],
|
|
1540
|
+
message: `duplicate check '${check}'`
|
|
1541
|
+
});
|
|
1542
|
+
seen.add(check);
|
|
1543
|
+
}
|
|
1544
|
+
});
|
|
1545
|
+
const nonNegativeCountSchema = z.number().int().min(0).max(Number.MAX_SAFE_INTEGER);
|
|
1546
|
+
const terminalBenchSchema = z.object({
|
|
1547
|
+
is_resolved: z.boolean().nullable(),
|
|
1548
|
+
failure_mode: z.string().min(1),
|
|
1549
|
+
parser_results: z.record(z.string().min(1), z.enum(["passed", "failed"])).nullable()
|
|
1550
|
+
}).passthrough();
|
|
1551
|
+
const directSweBenchSchema = z.object({
|
|
1552
|
+
resolved: z.boolean(),
|
|
1553
|
+
passed_tests: checkListSchema,
|
|
1554
|
+
failed_tests: checkListSchema
|
|
1555
|
+
}).passthrough();
|
|
1556
|
+
const nestedSweBenchCategorySchema = z.object({
|
|
1557
|
+
success: checkListSchema,
|
|
1558
|
+
failure: checkListSchema
|
|
1559
|
+
}).passthrough();
|
|
1560
|
+
const nestedSweBenchInstanceSchema = z.object({
|
|
1561
|
+
resolved: z.boolean(),
|
|
1562
|
+
tests_status: z.record(z.string().min(1), nestedSweBenchCategorySchema).refine((value) => Object.keys(value).length > 0, "must contain at least one test category")
|
|
1563
|
+
}).passthrough();
|
|
1564
|
+
const nestedSweBenchSchema = z.record(z.string().min(1), nestedSweBenchInstanceSchema).refine((value) => Object.keys(value).length > 0, "must contain at least one instance");
|
|
1565
|
+
const sweMultiCheckResultSchema = z.object({
|
|
1566
|
+
passed_count: nonNegativeCountSchema,
|
|
1567
|
+
failed_count: nonNegativeCountSchema,
|
|
1568
|
+
skipped_count: nonNegativeCountSchema,
|
|
1569
|
+
passed_tests: checkListSchema,
|
|
1570
|
+
failed_tests: checkListSchema,
|
|
1571
|
+
skipped_tests: checkListSchema
|
|
1572
|
+
}).passthrough();
|
|
1573
|
+
const sweMultiSchema = z.object({
|
|
1574
|
+
valid: z.boolean(),
|
|
1575
|
+
error_msg: z.string(),
|
|
1576
|
+
fix_patch_result: sweMultiCheckResultSchema
|
|
1577
|
+
}).passthrough();
|
|
1578
|
+
function parseVerificationOutcome(files) {
|
|
1579
|
+
if (files.length === 0) throw new Error("final verification outcome requires at least one result file");
|
|
1580
|
+
const sources = [];
|
|
1581
|
+
const passedChecks = /* @__PURE__ */ new Set();
|
|
1582
|
+
const failedChecks = /* @__PURE__ */ new Set();
|
|
1583
|
+
const unavailableReasons = /* @__PURE__ */ new Set();
|
|
1584
|
+
for (const file of files) {
|
|
1585
|
+
let value;
|
|
1586
|
+
try {
|
|
1587
|
+
value = JSON.parse(file.content);
|
|
1588
|
+
} catch (error) {
|
|
1589
|
+
throw new TypeError(`final verification result is not valid JSON: ${file.relativePath}: ${errorMessage(error)}`);
|
|
1590
|
+
}
|
|
1591
|
+
const parsed = parseResult(value, file.relativePath);
|
|
1592
|
+
sources.push({
|
|
1593
|
+
path: file.relativePath,
|
|
1594
|
+
format: parsed.format,
|
|
1595
|
+
status: parsed.status
|
|
1596
|
+
});
|
|
1597
|
+
if (parsed.reason) unavailableReasons.add(parsed.reason);
|
|
1598
|
+
for (const check of parsed.passedChecks) passedChecks.add(check);
|
|
1599
|
+
for (const check of parsed.failedChecks) failedChecks.add(check);
|
|
1600
|
+
}
|
|
1601
|
+
if (new Set(sources.map((source) => source.status)).size !== 1) throw new Error(`final verification result files disagree: ${sources.map((source) => `${source.path}=${source.status}`).join(", ")}`);
|
|
1602
|
+
if (unavailableReasons.size > 1) throw new Error(`final verification result files disagree on why the outcome is unavailable: ${[...unavailableReasons].join(", ")}`);
|
|
1603
|
+
const [unavailableReason] = unavailableReasons;
|
|
1604
|
+
const passed = [...passedChecks].sort();
|
|
1605
|
+
const failed = [...failedChecks].sort();
|
|
1606
|
+
assertDisjointChecks(passed, failed, "final verification result");
|
|
1607
|
+
return {
|
|
1608
|
+
status: sources[0].status,
|
|
1609
|
+
...unavailableReason ? { reason: unavailableReason } : {},
|
|
1610
|
+
sources,
|
|
1611
|
+
passedCheckCount: passed.length,
|
|
1612
|
+
failedCheckCount: failed.length,
|
|
1613
|
+
passedChecks: passed.slice(0, MAX_REPORTED_CHECKS),
|
|
1614
|
+
failedChecks: failed.slice(0, MAX_REPORTED_CHECKS)
|
|
1615
|
+
};
|
|
1616
|
+
}
|
|
1617
|
+
function parseResult(value, path) {
|
|
1618
|
+
const record = asRecord(value);
|
|
1619
|
+
if (!record) throw unsupported(path);
|
|
1620
|
+
const discriminators = [
|
|
1621
|
+
"is_resolved",
|
|
1622
|
+
"resolved",
|
|
1623
|
+
"valid"
|
|
1624
|
+
].filter((field) => Object.hasOwn(record, field));
|
|
1625
|
+
if (discriminators.length > 1) throw new TypeError(`final verification result is ambiguous: ${path}: found ${discriminators.join(", ")}`);
|
|
1626
|
+
if (discriminators[0] === "is_resolved") return parseTerminalBench(record, path);
|
|
1627
|
+
if (discriminators[0] === "resolved") return parseDirectSweBench(record, path);
|
|
1628
|
+
if (discriminators[0] === "valid") return parseSweMulti(record, path);
|
|
1629
|
+
if (Object.entries(record).some(([, candidate]) => {
|
|
1630
|
+
const nested = asRecord(candidate);
|
|
1631
|
+
return nested !== null && Object.hasOwn(nested, "resolved");
|
|
1632
|
+
})) return parseNestedSweBench(record, path);
|
|
1633
|
+
throw unsupported(path);
|
|
1634
|
+
}
|
|
1635
|
+
function parseTerminalBench(value, path) {
|
|
1636
|
+
const record = parseSchema(terminalBenchSchema, value, path, "Terminal-Bench");
|
|
1637
|
+
if (record.is_resolved === null) {
|
|
1638
|
+
if (record.parser_results !== null) throw malformed(path, "Terminal-Bench", "parser_results must be null when is_resolved is null");
|
|
1639
|
+
if (record.failure_mode === "unset") throw malformed(path, "Terminal-Bench", "failure_mode cannot be 'unset' when unresolved");
|
|
1640
|
+
return {
|
|
1641
|
+
format: "terminal-bench",
|
|
1642
|
+
status: "unavailable",
|
|
1643
|
+
reason: record.failure_mode === "parse_error" ? "result-parse-error" : "result-output-unavailable",
|
|
1644
|
+
passedChecks: [],
|
|
1645
|
+
failedChecks: []
|
|
1646
|
+
};
|
|
1647
|
+
}
|
|
1648
|
+
if (record.parser_results === null) throw malformed(path, "Terminal-Bench", "parser_results must be an object when is_resolved is boolean");
|
|
1649
|
+
const checks = stringStatusChecks(record.parser_results);
|
|
1650
|
+
if (checks.passedChecks.length + checks.failedChecks.length === 0) throw malformed(path, "Terminal-Bench", "parser_results must contain at least one check");
|
|
1651
|
+
assertOutcomeConsistency(record.is_resolved, checks, path, "Terminal-Bench is_resolved", true);
|
|
1652
|
+
return {
|
|
1653
|
+
format: "terminal-bench",
|
|
1654
|
+
status: status(record.is_resolved),
|
|
1655
|
+
...checks
|
|
1656
|
+
};
|
|
1657
|
+
}
|
|
1658
|
+
function parseDirectSweBench(value, path) {
|
|
1659
|
+
const record = parseSchema(directSweBenchSchema, value, path, "SWE-bench");
|
|
1660
|
+
const checks = {
|
|
1661
|
+
passedChecks: record.passed_tests,
|
|
1662
|
+
failedChecks: record.failed_tests
|
|
1663
|
+
};
|
|
1664
|
+
assertOutcomeConsistency(record.resolved, checks, path, "SWE-bench resolved");
|
|
1665
|
+
return {
|
|
1666
|
+
format: "swe-bench",
|
|
1667
|
+
status: status(record.resolved),
|
|
1668
|
+
...checks
|
|
1669
|
+
};
|
|
1670
|
+
}
|
|
1671
|
+
function parseNestedSweBench(value, path) {
|
|
1672
|
+
const record = parseSchema(nestedSweBenchSchema, value, path, "SWE-bench instance report");
|
|
1673
|
+
const instances = Object.entries(record);
|
|
1674
|
+
const statuses = new Set(instances.map(([, instance]) => status(instance.resolved)));
|
|
1675
|
+
if (statuses.size !== 1) throw new Error(`final verification report contains conflicting instance outcomes: ${path}: ${instances.map(([id, instance]) => `${id}=${status(instance.resolved)}`).join(", ")}`);
|
|
1676
|
+
const passedChecks = [];
|
|
1677
|
+
const failedChecks = [];
|
|
1678
|
+
for (const [instanceId, instance] of instances) {
|
|
1679
|
+
const checks = nestedSweBenchChecks(instanceId, instance.tests_status);
|
|
1680
|
+
assertOutcomeConsistency(instance.resolved, checks, path, `SWE-bench instance '${instanceId}' resolved`);
|
|
1681
|
+
passedChecks.push(...checks.passedChecks);
|
|
1682
|
+
failedChecks.push(...checks.failedChecks);
|
|
1683
|
+
}
|
|
1684
|
+
return {
|
|
1685
|
+
format: "swe-bench",
|
|
1686
|
+
status: [...statuses][0],
|
|
1687
|
+
passedChecks,
|
|
1688
|
+
failedChecks
|
|
1689
|
+
};
|
|
1690
|
+
}
|
|
1691
|
+
function parseSweMulti(value, path) {
|
|
1692
|
+
const record = parseSchema(sweMultiSchema, value, path, "SWE-Multi");
|
|
1693
|
+
const fix = record.fix_patch_result;
|
|
1694
|
+
assertCount(fix.passed_count, fix.passed_tests, "passed", path);
|
|
1695
|
+
assertCount(fix.failed_count, fix.failed_tests, "failed", path);
|
|
1696
|
+
assertCount(fix.skipped_count, fix.skipped_tests, "skipped", path);
|
|
1697
|
+
assertDisjointChecks(fix.passed_tests, fix.failed_tests, `SWE-Multi result ${path}`);
|
|
1698
|
+
assertDisjointChecks(fix.passed_tests, fix.skipped_tests, `SWE-Multi result ${path}`);
|
|
1699
|
+
assertDisjointChecks(fix.failed_tests, fix.skipped_tests, `SWE-Multi result ${path}`);
|
|
1700
|
+
const checks = {
|
|
1701
|
+
passedChecks: fix.passed_tests,
|
|
1702
|
+
failedChecks: fix.failed_tests
|
|
1703
|
+
};
|
|
1704
|
+
if (record.valid === false && isSweMultiOutputUnavailable(record)) return {
|
|
1705
|
+
format: "swe-multi",
|
|
1706
|
+
status: "unavailable",
|
|
1707
|
+
reason: "result-output-unavailable",
|
|
1708
|
+
...checks
|
|
1709
|
+
};
|
|
1710
|
+
assertOutcomeConsistency(record.valid, checks, path, "SWE-Multi valid");
|
|
1711
|
+
return {
|
|
1712
|
+
format: "swe-multi",
|
|
1713
|
+
status: status(record.valid),
|
|
1714
|
+
...checks
|
|
1715
|
+
};
|
|
1716
|
+
}
|
|
1717
|
+
function isSweMultiOutputUnavailable(record) {
|
|
1718
|
+
const fix = record.fix_patch_result;
|
|
1719
|
+
return (record.error_msg === SWE_MULTI_NO_TEST_RESULTS || record.error_msg.startsWith(`${SWE_MULTI_NO_TEST_RESULTS} `)) && fix.passed_count === 0 && fix.failed_count === 0 && fix.skipped_count === 0;
|
|
1720
|
+
}
|
|
1721
|
+
function stringStatusChecks(record) {
|
|
1722
|
+
const passedChecks = [];
|
|
1723
|
+
const failedChecks = [];
|
|
1724
|
+
for (const [name, result] of Object.entries(record)) if (result === "passed") passedChecks.push(name);
|
|
1725
|
+
else failedChecks.push(name);
|
|
1726
|
+
return {
|
|
1727
|
+
passedChecks,
|
|
1728
|
+
failedChecks
|
|
1729
|
+
};
|
|
1730
|
+
}
|
|
1731
|
+
function nestedSweBenchChecks(instanceId, testsStatus) {
|
|
1732
|
+
const passedChecks = [];
|
|
1733
|
+
const failedChecks = [];
|
|
1734
|
+
for (const [category, result] of Object.entries(testsStatus)) {
|
|
1735
|
+
passedChecks.push(...result.success.map((name) => `${instanceId}:${category}:${name}`));
|
|
1736
|
+
failedChecks.push(...result.failure.map((name) => `${instanceId}:${category}:${name}`));
|
|
1737
|
+
}
|
|
1738
|
+
return {
|
|
1739
|
+
passedChecks,
|
|
1740
|
+
failedChecks
|
|
1741
|
+
};
|
|
1742
|
+
}
|
|
1743
|
+
function assertOutcomeConsistency(passed, checks, path, field, requireFailedCheck = false) {
|
|
1744
|
+
assertDisjointChecks(checks.passedChecks, checks.failedChecks, `${field} in ${path}`);
|
|
1745
|
+
if (passed && checks.failedChecks.length > 0) throw malformed(path, field, `cannot be true while failed checks are reported: ${checks.failedChecks.join(", ")}`);
|
|
1746
|
+
if (passed && checks.passedChecks.length === 0) throw malformed(path, field, "cannot be true without at least one passed check");
|
|
1747
|
+
if (!passed && requireFailedCheck && checks.failedChecks.length === 0) throw malformed(path, field, "cannot be false without at least one failed check");
|
|
1748
|
+
}
|
|
1749
|
+
function assertCount(count, checks, kind, path) {
|
|
1750
|
+
if (count !== checks.length) throw malformed(path, "SWE-Multi", `${kind}_count=${count} does not match ${kind}_tests length ${checks.length}`);
|
|
1751
|
+
}
|
|
1752
|
+
function assertDisjointChecks(left, right, source) {
|
|
1753
|
+
const rightSet = new Set(right);
|
|
1754
|
+
const contradictions = [...new Set(left.filter((check) => rightSet.has(check)))].sort();
|
|
1755
|
+
if (contradictions.length > 0) throw new Error(`${source} marks checks as both passed and failed: ${contradictions.join(", ")}`);
|
|
1756
|
+
}
|
|
1757
|
+
function parseSchema(schema, value, path, format) {
|
|
1758
|
+
const parsed = schema.safeParse(value);
|
|
1759
|
+
if (parsed.success) return parsed.data;
|
|
1760
|
+
throw malformed(path, format, parsed.error.issues.map((issue) => `${issue.path.join(".") || "<root>"}: ${issue.message}`).join("; "));
|
|
1761
|
+
}
|
|
1762
|
+
function malformed(path, format, details) {
|
|
1763
|
+
return /* @__PURE__ */ new TypeError(`malformed ${format} verification result: ${path}: ${details}`);
|
|
1764
|
+
}
|
|
1765
|
+
function asRecord(value) {
|
|
1766
|
+
return typeof value === "object" && value !== null && !Array.isArray(value) ? value : null;
|
|
1767
|
+
}
|
|
1768
|
+
function status(value) {
|
|
1769
|
+
return value ? "passed" : "failed";
|
|
1770
|
+
}
|
|
1771
|
+
function unsupported(path) {
|
|
1772
|
+
return /* @__PURE__ */ new TypeError(`final verification result has no supported outcome field: ${path}; expected is_resolved, resolved, valid, or a SWE-bench instance report`);
|
|
1773
|
+
}
|
|
1774
|
+
function errorMessage(error) {
|
|
1775
|
+
return error instanceof Error ? error.message : String(error);
|
|
1776
|
+
}
|
|
1777
|
+
//#endregion
|
|
1778
|
+
//#region src/analyst/benchmark-verification-artifacts.ts
|
|
1779
|
+
const DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES = 8 * 1024 * 1024;
|
|
1780
|
+
const SEARCHED_ARTIFACTS = {
|
|
1781
|
+
"final-test-output": [
|
|
1782
|
+
"panes/post-test.txt",
|
|
1783
|
+
"sessions/tests.log",
|
|
1784
|
+
"test_output.txt"
|
|
1785
|
+
],
|
|
1786
|
+
"final-result": [
|
|
1787
|
+
"results.json",
|
|
1788
|
+
"result.json",
|
|
1789
|
+
"report.json",
|
|
1790
|
+
"*_result.json"
|
|
1791
|
+
],
|
|
1792
|
+
"final-metrics": ["*_metrics.json"]
|
|
1793
|
+
};
|
|
1794
|
+
const REQUIRED_ROLES = /* @__PURE__ */ new Set(["final-result"]);
|
|
1795
|
+
const UTF8 = new TextDecoder$1("utf-8", { fatal: true });
|
|
1796
|
+
async function loadCodeTraceVerificationArtifacts(options) {
|
|
1797
|
+
const maxBytes = positiveInteger(options.maxBytes ?? 8388608, "max verification artifact bytes");
|
|
1798
|
+
const sourceRelativePath = nonEmpty(options.row.source_relpath, `CodeTraceBench '${options.row.traj_id}' source_relpath`);
|
|
1799
|
+
const artifactRoot = await realpath(resolve(options.artifactDir));
|
|
1800
|
+
const caseDirectoriesSearched = [resolve(artifactRoot, options.row.traj_id, sourceRelativePath), resolve(artifactRoot, sourceRelativePath)].filter((path, index, paths) => paths.indexOf(path) === index);
|
|
1801
|
+
for (const path of caseDirectoriesSearched) assertContained(artifactRoot, path, sourceRelativePath);
|
|
1802
|
+
const existingCaseDirectories = /* @__PURE__ */ new Set();
|
|
1803
|
+
for (const path of caseDirectoriesSearched) try {
|
|
1804
|
+
const canonicalPath = await realpath(path);
|
|
1805
|
+
assertContained(artifactRoot, canonicalPath, sourceRelativePath);
|
|
1806
|
+
if (!(await stat(canonicalPath)).isDirectory()) throw new TypeError(`CodeTraceBench '${options.row.traj_id}' artifact case path is not a directory: ${canonicalPath}`);
|
|
1807
|
+
existingCaseDirectories.add(canonicalPath);
|
|
1808
|
+
} catch (error) {
|
|
1809
|
+
if (!isMissing(error)) throw error;
|
|
1810
|
+
}
|
|
1811
|
+
if (existingCaseDirectories.size === 0) return missingArtifacts(options.row.traj_id, caseDirectoriesSearched[0], caseDirectoriesSearched, maxBytes);
|
|
1812
|
+
if (existingCaseDirectories.size > 1) throw new Error(`CodeTraceBench '${options.row.traj_id}' artifact directory is ambiguous: ${[...existingCaseDirectories].join(", ")}`);
|
|
1813
|
+
const [caseDirectory] = existingCaseDirectories;
|
|
1814
|
+
const candidates = await artifactCandidates(caseDirectory);
|
|
1815
|
+
const files = [];
|
|
1816
|
+
let totalBytes = 0;
|
|
1817
|
+
for (const candidate of candidates) {
|
|
1818
|
+
const { bytes, canonicalPath } = await readArtifactSnapshot({
|
|
1819
|
+
artifactRoot,
|
|
1820
|
+
candidatePath: candidate.path,
|
|
1821
|
+
relativePath: candidate.relativePath,
|
|
1822
|
+
traceId: options.row.traj_id,
|
|
1823
|
+
totalBytes,
|
|
1824
|
+
maxBytes
|
|
1825
|
+
});
|
|
1826
|
+
totalBytes += bytes.byteLength;
|
|
1827
|
+
let content;
|
|
1828
|
+
try {
|
|
1829
|
+
content = UTF8.decode(bytes);
|
|
1830
|
+
} catch {
|
|
1831
|
+
throw new TypeError(`CodeTraceBench '${options.row.traj_id}' verification artifact is not UTF-8 text: ${canonicalPath}`);
|
|
1832
|
+
}
|
|
1833
|
+
if (!content.trim()) throw new Error(`CodeTraceBench '${options.row.traj_id}' verification artifact is empty: ${canonicalPath}`);
|
|
1834
|
+
const relativePath = candidate.relativePath;
|
|
1835
|
+
files.push({
|
|
1836
|
+
role: candidate.role,
|
|
1837
|
+
path: canonicalPath,
|
|
1838
|
+
relativePath,
|
|
1839
|
+
sha256: sha256Digest(bytes),
|
|
1840
|
+
bytes: bytes.byteLength,
|
|
1841
|
+
spanId: verificationSpanId(candidate.role, relativePath),
|
|
1842
|
+
content
|
|
1843
|
+
});
|
|
1844
|
+
}
|
|
1845
|
+
const roles = new Set(files.map((file) => file.role));
|
|
1846
|
+
const missingRoles = Object.keys(SEARCHED_ARTIFACTS).filter((role) => !roles.has(role));
|
|
1847
|
+
const hasFinalVerification = [...REQUIRED_ROLES].every((role) => roles.has(role));
|
|
1848
|
+
const outcome = hasFinalVerification ? loadVerificationOutcome(files.filter((file) => file.role === "final-result").map((file) => ({
|
|
1849
|
+
relativePath: file.relativePath,
|
|
1850
|
+
content: file.content
|
|
1851
|
+
})), options.row) : unavailableOutcome("missing-result");
|
|
1852
|
+
const outcomeSpanId = verificationOutcomeSpanId(options.row.traj_id, outcome);
|
|
1853
|
+
return {
|
|
1854
|
+
manifest: {
|
|
1855
|
+
traceId: options.row.traj_id,
|
|
1856
|
+
status: hasFinalVerification ? "present" : "missing",
|
|
1857
|
+
outcome,
|
|
1858
|
+
outcomeSpanId,
|
|
1859
|
+
caseDirectory,
|
|
1860
|
+
caseDirectoriesSearched,
|
|
1861
|
+
totalBytes,
|
|
1862
|
+
maxBytes,
|
|
1863
|
+
files: files.map(({ content: _content, ...file }) => file),
|
|
1864
|
+
missingRoles,
|
|
1865
|
+
searched: searchedArtifacts()
|
|
1866
|
+
},
|
|
1867
|
+
outcome,
|
|
1868
|
+
files
|
|
1869
|
+
};
|
|
1870
|
+
}
|
|
1871
|
+
function appendVerificationArtifactsToOtlp(otlpText, traceId, artifacts, afterTimestamp) {
|
|
1872
|
+
if (!otlpText.trim()) throw new Error(`trace '${traceId}' OTLP input is empty`);
|
|
1873
|
+
if (artifacts.manifest.traceId !== traceId) throw new Error(`verification artifacts for trace '${artifacts.manifest.traceId}' cannot be attached to '${traceId}'`);
|
|
1874
|
+
if (!artifacts.outcome || !artifacts.manifest.outcomeSpanId) throw new Error(`trace '${traceId}' has no final verification artifacts to attach`);
|
|
1875
|
+
const afterMs = Date.parse(afterTimestamp);
|
|
1876
|
+
if (!Number.isFinite(afterMs)) throw new TypeError(`trace '${traceId}' latest timestamp is invalid: ${afterTimestamp}`);
|
|
1877
|
+
const outcome = artifacts.outcome;
|
|
1878
|
+
const outcomeLine = JSON.stringify({
|
|
1879
|
+
trace_id: traceId,
|
|
1880
|
+
span_id: artifacts.manifest.outcomeSpanId,
|
|
1881
|
+
parent_span_id: null,
|
|
1882
|
+
name: `final verification outcome: ${outcome.status}`,
|
|
1883
|
+
start_time: timestampAfter(afterMs, 1, traceId),
|
|
1884
|
+
end_time: timestampAfter(afterMs, 2, traceId),
|
|
1885
|
+
status: { code: outcome.status === "passed" ? "STATUS_CODE_OK" : outcome.status === "failed" ? "STATUS_CODE_ERROR" : "STATUS_CODE_UNSET" },
|
|
1886
|
+
resource: { attributes: { "service.name": "agent-eval-public-benchmark" } },
|
|
1887
|
+
attributes: {
|
|
1888
|
+
"openinference.span.kind": "EVALUATOR",
|
|
1889
|
+
"benchmark.evidence.role": "final-verification",
|
|
1890
|
+
"benchmark.verification.outcome": outcome.status,
|
|
1891
|
+
"benchmark.verification.passed_check_count": outcome.passedCheckCount,
|
|
1892
|
+
"benchmark.verification.failed_check_count": outcome.failedCheckCount,
|
|
1893
|
+
"benchmark.verification.passed_checks": JSON.stringify(outcome.passedChecks),
|
|
1894
|
+
"benchmark.verification.failed_checks": JSON.stringify(outcome.failedChecks),
|
|
1895
|
+
"benchmark.verification.sources": JSON.stringify(outcome.sources),
|
|
1896
|
+
...outcome.reason ? { "benchmark.verification.reason": outcome.reason } : {},
|
|
1897
|
+
...outcome.parseError ? { "benchmark.verification.parse_error": JSON.stringify(outcome.parseError) } : {}
|
|
1898
|
+
}
|
|
1899
|
+
});
|
|
1900
|
+
const artifactLines = artifacts.files.filter((artifact) => artifact.role === "final-test-output").map((artifact, index) => JSON.stringify({
|
|
1901
|
+
trace_id: traceId,
|
|
1902
|
+
span_id: artifact.spanId,
|
|
1903
|
+
parent_span_id: null,
|
|
1904
|
+
name: `final verification artifact: ${artifact.relativePath}`,
|
|
1905
|
+
start_time: timestampAfter(afterMs, index * 2 + 3, traceId),
|
|
1906
|
+
end_time: timestampAfter(afterMs, index * 2 + 4, traceId),
|
|
1907
|
+
status: { code: "STATUS_CODE_UNSET" },
|
|
1908
|
+
resource: { attributes: { "service.name": "agent-eval-public-benchmark" } },
|
|
1909
|
+
attributes: {
|
|
1910
|
+
"openinference.span.kind": "EVALUATOR",
|
|
1911
|
+
"benchmark.evidence.role": "final-verification-artifact",
|
|
1912
|
+
"benchmark.verification.outcome": outcome.status,
|
|
1913
|
+
"artifact.role": artifact.role,
|
|
1914
|
+
"artifact.path": artifact.relativePath,
|
|
1915
|
+
"artifact.sha256": artifact.sha256,
|
|
1916
|
+
"artifact.bytes": artifact.bytes,
|
|
1917
|
+
"artifact.content": artifact.content
|
|
1918
|
+
}
|
|
1919
|
+
}));
|
|
1920
|
+
return `${otlpText.trimEnd()}\n${[outcomeLine, ...artifactLines].join("\n")}\n`;
|
|
1921
|
+
}
|
|
1922
|
+
function timestampAfter(afterMs, offsetMs, traceId) {
|
|
1923
|
+
const date = new Date(afterMs + offsetMs);
|
|
1924
|
+
if (!Number.isFinite(date.getTime())) throw new RangeError(`trace '${traceId}' cannot place final verification after its latest span`);
|
|
1925
|
+
return date.toISOString();
|
|
1926
|
+
}
|
|
1927
|
+
function sha256Digest(value) {
|
|
1928
|
+
return createHash("sha256").update(value).digest("hex");
|
|
1929
|
+
}
|
|
1930
|
+
async function readArtifactSnapshot(options) {
|
|
1931
|
+
const safeOpenFlags = constants.O_RDONLY | (process.platform === "win32" ? 0 : constants.O_NOFOLLOW | constants.O_NONBLOCK);
|
|
1932
|
+
let handle;
|
|
1933
|
+
try {
|
|
1934
|
+
handle = await open(options.candidatePath, safeOpenFlags);
|
|
1935
|
+
} catch (error) {
|
|
1936
|
+
if (isNodeError$1(error, "ELOOP")) throw new Error(`CodeTraceBench '${options.traceId}' verification artifact must not be a symbolic link: ${options.relativePath}`);
|
|
1937
|
+
throw error;
|
|
1938
|
+
}
|
|
1939
|
+
try {
|
|
1940
|
+
const before = await handle.stat();
|
|
1941
|
+
if (!before.isFile()) throw new TypeError(`CodeTraceBench '${options.traceId}' verification artifact is not a regular file: ${options.relativePath}`);
|
|
1942
|
+
const bytes = checkedFileSize(before.size, options);
|
|
1943
|
+
const descriptorPath = await openedDescriptorPath(handle.fd);
|
|
1944
|
+
if (descriptorPath !== null) assertContained(options.artifactRoot, descriptorPath, options.relativePath);
|
|
1945
|
+
const canonicalPath = await realpath(options.candidatePath);
|
|
1946
|
+
assertContained(options.artifactRoot, canonicalPath, options.relativePath);
|
|
1947
|
+
if (!sameFile(before, await stat(canonicalPath))) throw changedArtifact(options.traceId, options.relativePath);
|
|
1948
|
+
const content = Buffer.allocUnsafe(bytes);
|
|
1949
|
+
let offset = 0;
|
|
1950
|
+
while (offset < content.byteLength) {
|
|
1951
|
+
const { bytesRead } = await handle.read(content, offset, content.byteLength - offset, offset);
|
|
1952
|
+
if (bytesRead === 0) break;
|
|
1953
|
+
offset += bytesRead;
|
|
1954
|
+
}
|
|
1955
|
+
const eofProbe = Buffer.allocUnsafe(1);
|
|
1956
|
+
const { bytesRead: trailingBytes } = await handle.read(eofProbe, 0, eofProbe.byteLength, content.byteLength);
|
|
1957
|
+
const after = await handle.stat();
|
|
1958
|
+
if (offset !== content.byteLength || trailingBytes !== 0 || !sameSnapshot(before, after)) throw changedArtifact(options.traceId, options.relativePath);
|
|
1959
|
+
return {
|
|
1960
|
+
canonicalPath,
|
|
1961
|
+
bytes: content
|
|
1962
|
+
};
|
|
1963
|
+
} finally {
|
|
1964
|
+
await handle.close();
|
|
1965
|
+
}
|
|
1966
|
+
}
|
|
1967
|
+
function checkedFileSize(bytes, options) {
|
|
1968
|
+
if (!Number.isSafeInteger(bytes) || bytes < 0) throw new RangeError(`CodeTraceBench '${options.traceId}' verification artifact has an invalid byte size: ${options.relativePath}`);
|
|
1969
|
+
if (bytes > options.maxBytes) throw new RangeError(`CodeTraceBench '${options.traceId}' verification artifact '${options.relativePath}' requires ${bytes} bytes, over the ${options.maxBytes}-byte per-file limit`);
|
|
1970
|
+
if (options.totalBytes > options.maxBytes - bytes) throw new RangeError(`CodeTraceBench '${options.traceId}' verification artifacts require ${options.totalBytes + bytes} bytes, over the ${options.maxBytes}-byte cumulative limit`);
|
|
1971
|
+
return bytes;
|
|
1972
|
+
}
|
|
1973
|
+
async function openedDescriptorPath(fileDescriptor) {
|
|
1974
|
+
if (process.platform !== "linux") return null;
|
|
1975
|
+
try {
|
|
1976
|
+
return await realpath(`/proc/self/fd/${fileDescriptor}`);
|
|
1977
|
+
} catch (error) {
|
|
1978
|
+
if (isNodeError$1(error, "ENOENT") || isNodeError$1(error, "ENOTDIR") || isNodeError$1(error, "EACCES")) return null;
|
|
1979
|
+
throw error;
|
|
1980
|
+
}
|
|
1981
|
+
}
|
|
1982
|
+
function sameFile(left, right) {
|
|
1983
|
+
return left.isFile() && right.isFile() && left.dev === right.dev && left.ino === right.ino && left.size === right.size;
|
|
1984
|
+
}
|
|
1985
|
+
function sameSnapshot(left, right) {
|
|
1986
|
+
return sameFile(left, right) && left.mode === right.mode && left.mtimeMs === right.mtimeMs && left.ctimeMs === right.ctimeMs;
|
|
1987
|
+
}
|
|
1988
|
+
function changedArtifact(traceId, relativePath) {
|
|
1989
|
+
return /* @__PURE__ */ new Error(`CodeTraceBench '${traceId}' verification artifact changed while being read: ${relativePath}`);
|
|
1990
|
+
}
|
|
1991
|
+
async function artifactCandidates(caseDirectory) {
|
|
1992
|
+
const rootFiles = (await readdir(caseDirectory, { withFileTypes: true })).filter((entry) => entry.isFile() || entry.isSymbolicLink()).map((entry) => entry.name);
|
|
1993
|
+
const testOutput = await firstExisting(caseDirectory, SEARCHED_ARTIFACTS["final-test-output"]);
|
|
1994
|
+
const finalResults = [...SEARCHED_ARTIFACTS["final-result"].filter((name) => !name.includes("*")).map((name) => resolve(caseDirectory, name)), ...rootFiles.filter((name) => name.endsWith("_result.json")).map((name) => resolve(caseDirectory, name))];
|
|
1995
|
+
const finalMetrics = rootFiles.filter((name) => name.endsWith("_metrics.json")).map((name) => resolve(caseDirectory, name));
|
|
1996
|
+
const candidates = [
|
|
1997
|
+
...testOutput.map((path) => candidate("final-test-output", caseDirectory, path)),
|
|
1998
|
+
...(await existing(finalResults)).map((path) => candidate("final-result", caseDirectory, path)),
|
|
1999
|
+
...(await existing(finalMetrics)).map((path) => candidate("final-metrics", caseDirectory, path))
|
|
2000
|
+
];
|
|
2001
|
+
const seen = /* @__PURE__ */ new Set();
|
|
2002
|
+
return candidates.filter((entry) => {
|
|
2003
|
+
if (seen.has(entry.path)) return false;
|
|
2004
|
+
seen.add(entry.path);
|
|
2005
|
+
return true;
|
|
2006
|
+
}).sort((left, right) => artifactRoleOrder(left.role) - artifactRoleOrder(right.role) || left.relativePath.localeCompare(right.relativePath));
|
|
2007
|
+
}
|
|
2008
|
+
async function firstExisting(caseDirectory, candidates) {
|
|
2009
|
+
for (const relativePath of candidates) {
|
|
2010
|
+
const path = resolve(caseDirectory, relativePath);
|
|
2011
|
+
if (await isFile(path)) return [path];
|
|
2012
|
+
}
|
|
2013
|
+
return [];
|
|
2014
|
+
}
|
|
2015
|
+
async function existing(paths) {
|
|
2016
|
+
const out = [];
|
|
2017
|
+
for (const path of paths) if (await isFile(path)) out.push(path);
|
|
2018
|
+
return out;
|
|
2019
|
+
}
|
|
2020
|
+
async function isFile(path) {
|
|
2021
|
+
try {
|
|
2022
|
+
return (await stat(path)).isFile();
|
|
2023
|
+
} catch (error) {
|
|
2024
|
+
if (isMissing(error)) return false;
|
|
2025
|
+
throw error;
|
|
2026
|
+
}
|
|
2027
|
+
}
|
|
2028
|
+
function candidate(role, caseDirectory, path) {
|
|
2029
|
+
return {
|
|
2030
|
+
role,
|
|
2031
|
+
path,
|
|
2032
|
+
relativePath: slashRelative$1(caseDirectory, path)
|
|
2033
|
+
};
|
|
2034
|
+
}
|
|
2035
|
+
function missingArtifacts(traceId, caseDirectory, caseDirectoriesSearched, maxBytes) {
|
|
2036
|
+
const outcome = unavailableOutcome("missing-result");
|
|
2037
|
+
return {
|
|
2038
|
+
manifest: {
|
|
2039
|
+
traceId,
|
|
2040
|
+
status: "missing",
|
|
2041
|
+
outcome,
|
|
2042
|
+
outcomeSpanId: verificationOutcomeSpanId(traceId, outcome),
|
|
2043
|
+
caseDirectory,
|
|
2044
|
+
caseDirectoriesSearched,
|
|
2045
|
+
totalBytes: 0,
|
|
2046
|
+
maxBytes,
|
|
2047
|
+
files: [],
|
|
2048
|
+
missingRoles: Object.keys(SEARCHED_ARTIFACTS),
|
|
2049
|
+
searched: searchedArtifacts()
|
|
2050
|
+
},
|
|
2051
|
+
outcome,
|
|
2052
|
+
files: []
|
|
2053
|
+
};
|
|
2054
|
+
}
|
|
2055
|
+
function verificationOutcomeSpanId(traceId, outcome) {
|
|
2056
|
+
return `benchmark-verification-outcome-${sha256Digest(`${traceId}\u0000${JSON.stringify(outcome.sources)}\u0000${outcome.status}`).slice(0, 16)}`;
|
|
2057
|
+
}
|
|
2058
|
+
function unavailableOutcome(reason) {
|
|
2059
|
+
return {
|
|
2060
|
+
status: "unavailable",
|
|
2061
|
+
reason,
|
|
2062
|
+
sources: [],
|
|
2063
|
+
passedCheckCount: 0,
|
|
2064
|
+
failedCheckCount: 0,
|
|
2065
|
+
passedChecks: [],
|
|
2066
|
+
failedChecks: []
|
|
2067
|
+
};
|
|
2068
|
+
}
|
|
2069
|
+
function loadVerificationOutcome(files, row) {
|
|
2070
|
+
try {
|
|
2071
|
+
const outcome = parseVerificationOutcome(files);
|
|
2072
|
+
if (outcome.status === "unavailable" || typeof row.solved !== "boolean") return outcome;
|
|
2073
|
+
const labelStatus = row.solved ? "passed" : "failed";
|
|
2074
|
+
if (outcome.status === labelStatus) return outcome;
|
|
2075
|
+
return {
|
|
2076
|
+
...outcome,
|
|
2077
|
+
status: "unavailable",
|
|
2078
|
+
reason: "result-label-disagreement",
|
|
2079
|
+
parseError: {
|
|
2080
|
+
class: "ResultLabelDisagreementError",
|
|
2081
|
+
message: `CodeTraceBench '${row.traj_id}' solved=${row.solved} disagrees with parsed final verification status '${outcome.status}' from ${outcome.sources.map((source) => `${source.path}=${source.status}`).join(", ")}`
|
|
2082
|
+
}
|
|
2083
|
+
};
|
|
2084
|
+
} catch (error) {
|
|
2085
|
+
return {
|
|
2086
|
+
...unavailableOutcome("result-parse-error"),
|
|
2087
|
+
parseError: {
|
|
2088
|
+
class: error instanceof Error ? error.constructor.name : "Error",
|
|
2089
|
+
message: error instanceof Error ? error.message : String(error)
|
|
2090
|
+
}
|
|
2091
|
+
};
|
|
2092
|
+
}
|
|
2093
|
+
}
|
|
2094
|
+
function verificationSpanId(role, relativePath) {
|
|
2095
|
+
return `benchmark-verification-${sha256Digest(`${role}\u0000${relativePath}`).slice(0, 16)}`;
|
|
2096
|
+
}
|
|
2097
|
+
function artifactRoleOrder(role) {
|
|
2098
|
+
return role === "final-test-output" ? 0 : role === "final-result" ? 1 : 2;
|
|
2099
|
+
}
|
|
2100
|
+
function assertContained(root, candidate, source) {
|
|
2101
|
+
const rel = relative(root, candidate);
|
|
2102
|
+
if (isAbsolute(rel) || rel === ".." || rel.startsWith(`..${sep}`)) throw new Error(`verification artifact path escapes --artifact-dir: ${source}`);
|
|
2103
|
+
}
|
|
2104
|
+
function searchedArtifacts() {
|
|
2105
|
+
return {
|
|
2106
|
+
"final-test-output": [...SEARCHED_ARTIFACTS["final-test-output"]],
|
|
2107
|
+
"final-result": [...SEARCHED_ARTIFACTS["final-result"]],
|
|
2108
|
+
"final-metrics": [...SEARCHED_ARTIFACTS["final-metrics"]]
|
|
2109
|
+
};
|
|
2110
|
+
}
|
|
2111
|
+
function slashRelative$1(root, path) {
|
|
2112
|
+
return relative(root, path).split(sep).join("/");
|
|
2113
|
+
}
|
|
2114
|
+
function nonEmpty(value, field) {
|
|
2115
|
+
if (typeof value !== "string" || !value.trim()) throw new TypeError(`${field} must be a non-empty string`);
|
|
2116
|
+
return value.trim();
|
|
2117
|
+
}
|
|
2118
|
+
function positiveInteger(value, field) {
|
|
2119
|
+
if (!Number.isSafeInteger(value) || value < 1) throw new RangeError(`${field} must be a positive safe integer`);
|
|
2120
|
+
return value;
|
|
2121
|
+
}
|
|
2122
|
+
function isMissing(error) {
|
|
2123
|
+
return isNodeError$1(error, "ENOENT");
|
|
2124
|
+
}
|
|
2125
|
+
function isNodeError$1(error, code) {
|
|
2126
|
+
return error instanceof Error && "code" in error && error.code === code;
|
|
2127
|
+
}
|
|
2128
|
+
//#endregion
|
|
2129
|
+
//#region src/analyst/benchmark-public-data.ts
|
|
2130
|
+
const DEFAULT_MAX_PUBLIC_BENCHMARK_LABEL_BYTES = 256 * 1024 * 1024;
|
|
2131
|
+
const INPUT_OPEN_FLAGS = constants.O_RDONLY | (constants.O_NOFOLLOW ?? 0);
|
|
2132
|
+
async function loadPublicBenchmarkRows(path) {
|
|
2133
|
+
return parsePublicBenchmarkRows((await readImmutableInputSnapshot(resolve(path), DEFAULT_MAX_PUBLIC_BENCHMARK_LABEL_BYTES)).text, path);
|
|
2134
|
+
}
|
|
2135
|
+
function parsePublicBenchmarkRows(text, path) {
|
|
2136
|
+
const trimmed = text.trim();
|
|
2137
|
+
if (!trimmed) throw new Error(`public analyst benchmark dataset is empty: ${path}`);
|
|
2138
|
+
let parsed;
|
|
2139
|
+
try {
|
|
2140
|
+
parsed = JSON.parse(trimmed);
|
|
2141
|
+
} catch {
|
|
2142
|
+
return parseJsonl(trimmed, path);
|
|
2143
|
+
}
|
|
2144
|
+
if (Array.isArray(parsed)) return records(parsed, path);
|
|
2145
|
+
if (isRecord(parsed) && Array.isArray(parsed.data)) return records(parsed.data, `${path}.data`);
|
|
2146
|
+
if (isRecord(parsed) && Array.isArray(parsed.cases)) return records(parsed.cases, `${path}.cases`);
|
|
2147
|
+
if (isRecord(parsed)) return [parsed];
|
|
2148
|
+
throw new TypeError(`public analyst benchmark dataset must contain JSON objects: ${path}`);
|
|
2149
|
+
}
|
|
2150
|
+
function selectPublicBenchmarkRows(dataset, rows, options) {
|
|
2151
|
+
positiveSafeInteger(options.limit, "limit");
|
|
2152
|
+
safeInteger(options.seed, "seed");
|
|
2153
|
+
if (rows.length === 0) throw new Error("public analyst benchmark dataset has no rows");
|
|
2154
|
+
const byId = /* @__PURE__ */ new Map();
|
|
2155
|
+
for (const row of rows) {
|
|
2156
|
+
const id = publicBenchmarkRowId(dataset, row);
|
|
2157
|
+
if (byId.has(id)) throw new Error(`public analyst benchmark dataset repeats trajectory id '${id}'`);
|
|
2158
|
+
byId.set(id, row);
|
|
2159
|
+
}
|
|
2160
|
+
return [...byId].sort(([left], [right]) => selectionKey(options.seed, left).localeCompare(selectionKey(options.seed, right)) || left.localeCompare(right)).slice(0, Math.min(options.limit, byId.size)).map(([, row]) => row);
|
|
2161
|
+
}
|
|
2162
|
+
function publicBenchmarkDistributions(dataset, rows) {
|
|
2163
|
+
const values = {
|
|
2164
|
+
class: [],
|
|
2165
|
+
agent: [],
|
|
2166
|
+
model: [],
|
|
2167
|
+
difficulty: [],
|
|
2168
|
+
solved: []
|
|
2169
|
+
};
|
|
2170
|
+
for (const row of rows) {
|
|
2171
|
+
const benchmarkCase = dataset === "agentrx" ? agentRxBenchmarkCase(row, void 0) : codeTraceBenchCase(row, void 0);
|
|
2172
|
+
values.class.push(dataset === "codetracebench" ? benchmarkCase.expectedIssues.length > 0 ? "positive" : row.solved === true ? "trusted-negative" : row.solved === false ? "unlabeled-failure" : "unlabeled-unknown" : benchmarkCase.expectedIssues[0]?.areas?.[0]);
|
|
2173
|
+
values.agent.push(scalarDistributionValue(row.agent) ?? (dataset === "agentrx" ? rootAgent(row) : void 0));
|
|
2174
|
+
values.model.push(scalarDistributionValue(row.model));
|
|
2175
|
+
values.difficulty.push(scalarDistributionValue(row.difficulty));
|
|
2176
|
+
values.solved.push(scalarDistributionValue(row.solved));
|
|
2177
|
+
}
|
|
2178
|
+
return {
|
|
2179
|
+
class: valueDistribution(values.class),
|
|
2180
|
+
agent: valueDistribution(values.agent),
|
|
2181
|
+
model: valueDistribution(values.model),
|
|
2182
|
+
difficulty: valueDistribution(values.difficulty),
|
|
2183
|
+
solved: valueDistribution(values.solved)
|
|
2184
|
+
};
|
|
2185
|
+
}
|
|
2186
|
+
function publicBenchmarkSelectionReport(dataset, source, selected, seed) {
|
|
2187
|
+
const census = source.length === selected.length;
|
|
2188
|
+
return {
|
|
2189
|
+
method: census ? "census" : "deterministic-hash",
|
|
2190
|
+
seed,
|
|
2191
|
+
sourceCount: source.length,
|
|
2192
|
+
selectedCount: selected.length,
|
|
2193
|
+
stratified: false,
|
|
2194
|
+
representativeOfInput: census,
|
|
2195
|
+
source: publicBenchmarkDistributions(dataset, source),
|
|
2196
|
+
selected: publicBenchmarkDistributions(dataset, selected)
|
|
2197
|
+
};
|
|
2198
|
+
}
|
|
2199
|
+
async function preparePublicAnalystBenchmark(options) {
|
|
2200
|
+
const labelsPath = resolve(options.labelsPath);
|
|
2201
|
+
const traceRoot = resolve(options.traceDir);
|
|
2202
|
+
const artifactRoot = options.artifactDir ? resolve(options.artifactDir) : void 0;
|
|
2203
|
+
const labelSnapshot = await readImmutableInputSnapshot(labelsPath, DEFAULT_MAX_PUBLIC_BENCHMARK_LABEL_BYTES);
|
|
2204
|
+
const rows = parsePublicBenchmarkRows(labelSnapshot.text, labelsPath);
|
|
2205
|
+
const selected = selectPublicBenchmarkRows(options.dataset, rows, {
|
|
2206
|
+
limit: options.limit,
|
|
2207
|
+
seed: options.seed
|
|
2208
|
+
});
|
|
2209
|
+
const stores = await indexSelectedSingleTraceFiles(traceRoot, new Set(selected.map((row) => publicBenchmarkRowId(options.dataset, row))));
|
|
2210
|
+
const resolver = traceStoreEvidenceResolver((input) => {
|
|
2211
|
+
if (!input.traceStore) throw new Error("prepared benchmark case has no trace store");
|
|
2212
|
+
return input.traceStore;
|
|
2213
|
+
});
|
|
2214
|
+
const traceFiles = [];
|
|
2215
|
+
const verificationArtifacts = [];
|
|
2216
|
+
const cases = [];
|
|
2217
|
+
for (const row of selected) {
|
|
2218
|
+
const trajectoryId = publicBenchmarkRowId(options.dataset, row);
|
|
2219
|
+
const indexed = stores.get(trajectoryId);
|
|
2220
|
+
if (!indexed) throw new Error(`public analyst benchmark trace directory has no single-trace OTLP JSONL for '${trajectoryId}'`);
|
|
2221
|
+
let modelVisibleOtlp = indexed.text;
|
|
2222
|
+
let traceStore = indexed.store;
|
|
2223
|
+
let artifactDir;
|
|
2224
|
+
let verificationManifest;
|
|
2225
|
+
if (options.dataset === "codetracebench") {
|
|
2226
|
+
if (!options.artifactDir?.trim()) throw new Error("--artifact-dir is required for CodeTraceBench so final verification evidence is not omitted");
|
|
2227
|
+
const artifacts = await loadCodeTraceVerificationArtifacts({
|
|
2228
|
+
artifactDir: options.artifactDir,
|
|
2229
|
+
row,
|
|
2230
|
+
maxBytes: options.maxArtifactBytes ?? 8388608
|
|
2231
|
+
});
|
|
2232
|
+
for (const artifact of artifacts.files) assertNoBenchmarkLabelsInArtifact({
|
|
2233
|
+
traceId: trajectoryId,
|
|
2234
|
+
relativePath: artifact.relativePath,
|
|
2235
|
+
content: artifact.content
|
|
2236
|
+
});
|
|
2237
|
+
verificationManifest = shareableVerificationManifest(artifacts.manifest, artifactRoot ?? resolve(options.artifactDir));
|
|
2238
|
+
const collisions = await indexed.store.hasSpans({
|
|
2239
|
+
trace_id: trajectoryId,
|
|
2240
|
+
span_ids: [...artifacts.manifest.files.map((file) => file.spanId), artifacts.manifest.outcomeSpanId]
|
|
2241
|
+
});
|
|
2242
|
+
if (collisions.length > 0) throw new Error(`CodeTraceBench '${trajectoryId}' trace already contains benchmark verification span '${collisions[0]}'`);
|
|
2243
|
+
modelVisibleOtlp = appendVerificationArtifactsToOtlp(indexed.text, trajectoryId, artifacts, indexed.latestTimestamp);
|
|
2244
|
+
traceStore = otlpTextToTraceAnalysisStore(modelVisibleOtlp);
|
|
2245
|
+
artifactDir = artifacts.manifest.status === "present" ? artifacts.manifest.caseDirectory : void 0;
|
|
2246
|
+
verificationArtifacts.push(verificationManifest);
|
|
2247
|
+
}
|
|
2248
|
+
const labelLeakScan = assertNoBenchmarkLabelsInTrace({
|
|
2249
|
+
traceId: trajectoryId,
|
|
2250
|
+
otlpText: modelVisibleOtlp
|
|
2251
|
+
});
|
|
2252
|
+
const input = {
|
|
2253
|
+
traceStore,
|
|
2254
|
+
artifactDir
|
|
2255
|
+
};
|
|
2256
|
+
const benchmarkCase = options.dataset === "agentrx" ? agentRxBenchmarkCase(row, input, { stepCount: indexed.stepCount }) : codeTraceBenchCase(row, input);
|
|
2257
|
+
for (const evidence of benchmarkCase.labeledEvidence ?? []) if (!await resolver({
|
|
2258
|
+
caseId: benchmarkCase.id,
|
|
2259
|
+
caseInput: input,
|
|
2260
|
+
evidence: {
|
|
2261
|
+
kind: evidence.kind ?? "span",
|
|
2262
|
+
uri: evidence.uri
|
|
2263
|
+
}
|
|
2264
|
+
})) throw new Error(`${benchmarkCase.id}: missing labeled span ${spanIdFromEvidence(evidence.uri) ?? evidence.uri} in ${indexed.path}`);
|
|
2265
|
+
cases.push({
|
|
2266
|
+
...benchmarkCase,
|
|
2267
|
+
metadata: {
|
|
2268
|
+
...benchmarkCase.metadata,
|
|
2269
|
+
traceFileRelativePath: slashRelative(traceRoot, indexed.path),
|
|
2270
|
+
traceFileSha256: indexed.sha256,
|
|
2271
|
+
labelLeakScan,
|
|
2272
|
+
...verificationManifest ? { verificationArtifacts: verificationManifest } : {}
|
|
2273
|
+
}
|
|
2274
|
+
});
|
|
2275
|
+
traceFiles.push({
|
|
2276
|
+
traceId: trajectoryId,
|
|
2277
|
+
relativePath: slashRelative(traceRoot, indexed.path),
|
|
2278
|
+
sha256: indexed.sha256
|
|
2279
|
+
});
|
|
2280
|
+
}
|
|
2281
|
+
return {
|
|
2282
|
+
cases,
|
|
2283
|
+
sourceRowCount: rows.length,
|
|
2284
|
+
selectedCaseIds: cases.map((testCase) => testCase.id),
|
|
2285
|
+
labelsSha256: labelSnapshot.sha256,
|
|
2286
|
+
traceFiles,
|
|
2287
|
+
verificationArtifacts,
|
|
2288
|
+
selection: publicBenchmarkSelectionReport(options.dataset, rows, selected, options.seed)
|
|
2289
|
+
};
|
|
2290
|
+
}
|
|
2291
|
+
async function indexSelectedSingleTraceFiles(traceDir, selectedTraceIds) {
|
|
2292
|
+
if (selectedTraceIds.size === 0) throw new Error("public analyst benchmark selected no trace ids");
|
|
2293
|
+
const files = (await readdir(traceDir, { withFileTypes: true })).filter((entry) => entry.isFile() && entry.name.endsWith(".jsonl")).map((entry) => resolve(traceDir, entry.name)).sort();
|
|
2294
|
+
if (files.length === 0) throw new Error(`public analyst benchmark trace directory has no JSONL files: ${traceDir}`);
|
|
2295
|
+
const indexed = /* @__PURE__ */ new Map();
|
|
2296
|
+
for (const path of files) {
|
|
2297
|
+
const snapshot = await readImmutableInputSnapshot(path, DEFAULT_MAX_TRACE_FILE_BYTES);
|
|
2298
|
+
const store = createOtlpBufferTraceStore(snapshot.bytes);
|
|
2299
|
+
const overview = await store.getOverview();
|
|
2300
|
+
if (overview.total_traces !== 1 || overview.sample_trace_ids.length !== 1) throw new Error(`public analyst benchmark trace file must contain exactly one trace: ${path} contains ${overview.total_traces}`);
|
|
2301
|
+
if (!overview.time_range) throw new Error(`public analyst benchmark trace file has no valid timestamps: ${path}`);
|
|
2302
|
+
const traceId = overview.sample_trace_ids[0];
|
|
2303
|
+
if (!selectedTraceIds.has(traceId)) continue;
|
|
2304
|
+
if (indexed.has(traceId)) throw new Error(`public analyst benchmark trace id '${traceId}' appears in multiple files`);
|
|
2305
|
+
indexed.set(traceId, {
|
|
2306
|
+
path,
|
|
2307
|
+
sha256: snapshot.sha256,
|
|
2308
|
+
store,
|
|
2309
|
+
latestTimestamp: overview.time_range.latest,
|
|
2310
|
+
text: snapshot.text,
|
|
2311
|
+
stepCount: traceStepCount(snapshot.text, path)
|
|
2312
|
+
});
|
|
2313
|
+
}
|
|
2314
|
+
return indexed;
|
|
2315
|
+
}
|
|
2316
|
+
async function readImmutableInputSnapshot(path, maxBytes) {
|
|
2317
|
+
if (!Number.isSafeInteger(maxBytes) || maxBytes < 1) throw new RangeError("benchmark input maxBytes must be a positive safe integer");
|
|
2318
|
+
const handle = await open(path, INPUT_OPEN_FLAGS);
|
|
2319
|
+
try {
|
|
2320
|
+
return await readImmutableInputHandle(handle, path, maxBytes);
|
|
2321
|
+
} finally {
|
|
2322
|
+
await handle.close();
|
|
2323
|
+
}
|
|
2324
|
+
}
|
|
2325
|
+
async function readImmutableInputHandle(handle, path, maxBytes) {
|
|
2326
|
+
const before = await handle.stat({ bigint: true });
|
|
2327
|
+
if (!before.isFile()) throw new TypeError(`public analyst benchmark input must be a regular file: ${path}`);
|
|
2328
|
+
if (before.size > BigInt(maxBytes)) throw new RangeError(`public analyst benchmark input exceeds ${maxBytes} bytes: ${path} has ${before.size}`);
|
|
2329
|
+
const size = Number(before.size);
|
|
2330
|
+
const bytes = Buffer.allocUnsafe(size);
|
|
2331
|
+
let offset = 0;
|
|
2332
|
+
while (offset < size) {
|
|
2333
|
+
const result = await handle.read(bytes, offset, size - offset, offset);
|
|
2334
|
+
if (result.bytesRead === 0) throw new Error(`public analyst benchmark input changed while being read: ${path}`);
|
|
2335
|
+
offset += result.bytesRead;
|
|
2336
|
+
}
|
|
2337
|
+
const overflow = Buffer.allocUnsafe(1);
|
|
2338
|
+
const extra = await handle.read(overflow, 0, 1, size);
|
|
2339
|
+
const after = await handle.stat({ bigint: true });
|
|
2340
|
+
if (extra.bytesRead !== 0 || before.dev !== after.dev || before.ino !== after.ino || before.size !== after.size || before.mtimeNs !== after.mtimeNs || before.ctimeNs !== after.ctimeNs) throw new Error(`public analyst benchmark input changed while being read: ${path}`);
|
|
2341
|
+
let text;
|
|
2342
|
+
try {
|
|
2343
|
+
text = new TextDecoder("utf-8", { fatal: true }).decode(bytes);
|
|
2344
|
+
} catch (error) {
|
|
2345
|
+
throw new TypeError(`public analyst benchmark input is not valid UTF-8: ${path}: ${error instanceof Error ? error.message : String(error)}`);
|
|
2346
|
+
}
|
|
2347
|
+
return Object.freeze({
|
|
2348
|
+
bytes,
|
|
2349
|
+
sha256: sha256Digest(bytes),
|
|
2350
|
+
text
|
|
2351
|
+
});
|
|
2352
|
+
}
|
|
2353
|
+
function traceStepCount(text, path) {
|
|
2354
|
+
const steps = parseJsonl(text, path).map((row) => row.span_id).filter((spanId) => typeof spanId === "string").map((spanId) => /^step-(\d+)$/.exec(spanId)?.[1]).filter((step) => step !== void 0).map(Number).filter((step) => Number.isSafeInteger(step) && step > 0);
|
|
2355
|
+
if (steps.length === 0) throw new Error(`public analyst benchmark trace has no step-<n> spans: ${path}`);
|
|
2356
|
+
return Math.max(...steps);
|
|
2357
|
+
}
|
|
2358
|
+
function shareableVerificationManifest(manifest, artifactRoot) {
|
|
2359
|
+
return {
|
|
2360
|
+
...manifest,
|
|
2361
|
+
caseDirectory: slashRelative(artifactRoot, manifest.caseDirectory),
|
|
2362
|
+
caseDirectoriesSearched: manifest.caseDirectoriesSearched.map((path) => slashRelative(artifactRoot, path)),
|
|
2363
|
+
files: manifest.files.map((file) => ({
|
|
2364
|
+
...file,
|
|
2365
|
+
path: slashRelative(artifactRoot, file.path)
|
|
2366
|
+
}))
|
|
2367
|
+
};
|
|
2368
|
+
}
|
|
2369
|
+
function slashRelative(root, path) {
|
|
2370
|
+
const value = relative(root, path);
|
|
2371
|
+
if (!value || value === ".." || value.startsWith(`..${sep}`)) {
|
|
2372
|
+
if (!value) return ".";
|
|
2373
|
+
throw new Error(`benchmark artifact path escapes its declared root: ${path}`);
|
|
2374
|
+
}
|
|
2375
|
+
return value.replaceAll("\\", "/");
|
|
2376
|
+
}
|
|
2377
|
+
function parseJsonl(text, path) {
|
|
2378
|
+
const rows = [];
|
|
2379
|
+
for (const [index, line] of text.split(/\r?\n/).entries()) {
|
|
2380
|
+
const trimmed = line.trim();
|
|
2381
|
+
if (!trimmed) continue;
|
|
2382
|
+
let parsed;
|
|
2383
|
+
try {
|
|
2384
|
+
parsed = JSON.parse(trimmed);
|
|
2385
|
+
} catch (error) {
|
|
2386
|
+
throw new Error(`${path}:${index + 1}: invalid JSON: ${error instanceof Error ? error.message : String(error)}`);
|
|
2387
|
+
}
|
|
2388
|
+
if (!isRecord(parsed)) throw new TypeError(`${path}:${index + 1}: dataset row must be a JSON object`);
|
|
2389
|
+
rows.push(parsed);
|
|
2390
|
+
}
|
|
2391
|
+
if (rows.length === 0) throw new Error(`public analyst benchmark dataset is empty: ${path}`);
|
|
2392
|
+
return rows;
|
|
2393
|
+
}
|
|
2394
|
+
function records(values, path) {
|
|
2395
|
+
return values.map((value, index) => {
|
|
2396
|
+
if (!isRecord(value)) throw new TypeError(`${path}[${index}] must be a JSON object`);
|
|
2397
|
+
return value;
|
|
2398
|
+
});
|
|
2399
|
+
}
|
|
2400
|
+
function publicBenchmarkRowId(dataset, row) {
|
|
2401
|
+
const value = dataset === "agentrx" ? row.trajectory_id : row.traj_id;
|
|
2402
|
+
if (typeof value !== "string" && typeof value !== "number" || !String(value).trim()) throw new TypeError(`${dataset} dataset row requires a non-empty ${dataset === "agentrx" ? "trajectory_id" : "traj_id"}`);
|
|
2403
|
+
return String(value);
|
|
2404
|
+
}
|
|
2405
|
+
function spanIdFromEvidence(uri) {
|
|
2406
|
+
const match = /\/span\/([^/]+)$/.exec(uri);
|
|
2407
|
+
return match?.[1] ? decodeURIComponent(match[1]) : null;
|
|
2408
|
+
}
|
|
2409
|
+
function selectionKey(seed, id) {
|
|
2410
|
+
return sha256Digest(`${seed}\u0000${id}`);
|
|
2411
|
+
}
|
|
2412
|
+
function valueDistribution(values) {
|
|
2413
|
+
const counts = /* @__PURE__ */ new Map();
|
|
2414
|
+
let missing = 0;
|
|
2415
|
+
for (const value of values) {
|
|
2416
|
+
if (value === void 0) {
|
|
2417
|
+
missing += 1;
|
|
2418
|
+
continue;
|
|
2419
|
+
}
|
|
2420
|
+
counts.set(value, (counts.get(value) ?? 0) + 1);
|
|
2421
|
+
}
|
|
2422
|
+
return {
|
|
2423
|
+
total: values.length,
|
|
2424
|
+
missing,
|
|
2425
|
+
counts: Object.fromEntries([...counts].sort(([left], [right]) => left.localeCompare(right)))
|
|
2426
|
+
};
|
|
2427
|
+
}
|
|
2428
|
+
function scalarDistributionValue(value) {
|
|
2429
|
+
if (typeof value === "string") return value.trim() || void 0;
|
|
2430
|
+
if (typeof value === "number" || typeof value === "boolean") return String(value);
|
|
2431
|
+
}
|
|
2432
|
+
function rootAgent(row) {
|
|
2433
|
+
const rootCauseId = row.root_cause_failure_id ?? row.root_cause?.failure_id;
|
|
2434
|
+
return scalarDistributionValue(row.failures.find((failure) => String(failure.failure_id) === String(rootCauseId))?.failed_agent);
|
|
2435
|
+
}
|
|
2436
|
+
//#endregion
|
|
2437
|
+
//#region src/analyst/benchmark-public-errors.ts
|
|
2438
|
+
function publicBenchmarkError(error, secrets = []) {
|
|
2439
|
+
if (error instanceof LlmCallError) return {
|
|
2440
|
+
class: "LlmCallError",
|
|
2441
|
+
code: error.code,
|
|
2442
|
+
status: error.status,
|
|
2443
|
+
message: `Provider request failed with HTTP ${error.status}.`
|
|
2444
|
+
};
|
|
2445
|
+
if (error instanceof LlmResponseError) return {
|
|
2446
|
+
class: "LlmResponseError",
|
|
2447
|
+
code: error.code,
|
|
2448
|
+
message: "Provider response did not satisfy the structured output contract."
|
|
2449
|
+
};
|
|
2450
|
+
if (error instanceof z.ZodError) return {
|
|
2451
|
+
class: "ModelOutputValidationError",
|
|
2452
|
+
message: "Provider response did not match the benchmark output schema."
|
|
2453
|
+
};
|
|
2454
|
+
if (error instanceof SyntaxError) return {
|
|
2455
|
+
class: "ModelOutputParseError",
|
|
2456
|
+
message: "Provider response was not valid JSON."
|
|
2457
|
+
};
|
|
2458
|
+
if (error instanceof CostCeilingReachedError || error instanceof CostAccountingIncompleteError || error instanceof CostReservationExceededError) return {
|
|
2459
|
+
class: error.constructor.name,
|
|
2460
|
+
code: error.code,
|
|
2461
|
+
message: redactSensitiveText(error.message, secrets)
|
|
2462
|
+
};
|
|
2463
|
+
if (error instanceof Error && error.name === "AbortError") return {
|
|
2464
|
+
class: "ProviderTimeoutError",
|
|
2465
|
+
message: "Provider request timed out."
|
|
2466
|
+
};
|
|
2467
|
+
if (error instanceof Error && /(?:assistant steps?|finding evidence|selected missing|selected unavailable|no readable spans|requires a trace store)/i.test(error.message)) return {
|
|
2468
|
+
class: "BenchmarkEvidenceError",
|
|
2469
|
+
message: redactSensitiveText(error.message, secrets)
|
|
2470
|
+
};
|
|
2471
|
+
if (error instanceof AgentEvalError) return {
|
|
2472
|
+
class: error.constructor.name,
|
|
2473
|
+
code: error.code,
|
|
2474
|
+
message: redactSensitiveText(error.message, secrets)
|
|
2475
|
+
};
|
|
2476
|
+
if (error instanceof Error) return {
|
|
2477
|
+
class: error.constructor.name || "Error",
|
|
2478
|
+
message: redactSensitiveText(error.message, secrets)
|
|
2479
|
+
};
|
|
2480
|
+
return {
|
|
2481
|
+
class: "Error",
|
|
2482
|
+
message: "Benchmark analyst execution failed."
|
|
2483
|
+
};
|
|
2484
|
+
}
|
|
2485
|
+
function redactSensitiveText(value, secrets) {
|
|
2486
|
+
let redacted = value;
|
|
2487
|
+
for (const secret of secrets) if (secret) redacted = redacted.replaceAll(secret, "[REDACTED]");
|
|
2488
|
+
return redacted.replace(/\bBearer\s+[^\s"',;]+/gi, "Bearer [REDACTED]").replace(/\b(api[_-]?key|access[_-]?token|refresh[_-]?token|password|secret)\b\s*[:=]\s*[^\s"',;]+/gi, "$1=[REDACTED]").slice(0, 500);
|
|
2489
|
+
}
|
|
2490
|
+
//#endregion
|
|
2491
|
+
//#region src/analyst/benchmark-response-cache.ts
|
|
2492
|
+
const SHA256 = /^[a-f0-9]{64}$/;
|
|
2493
|
+
const CostReceiptInputSchema = z.object({
|
|
2494
|
+
model: z.string().min(1),
|
|
2495
|
+
inputTokens: z.number().int().nonnegative(),
|
|
2496
|
+
outputTokens: z.number().int().nonnegative(),
|
|
2497
|
+
reasoningTokens: z.number().int().nonnegative().optional(),
|
|
2498
|
+
cachedTokens: z.number().int().nonnegative().optional(),
|
|
2499
|
+
cacheWriteTokens: z.number().int().nonnegative().optional(),
|
|
2500
|
+
customTokenPricing: z.object({
|
|
2501
|
+
inputUsdPerMillion: z.number().nonnegative(),
|
|
2502
|
+
cachedInputUsdPerMillion: z.number().nonnegative().optional(),
|
|
2503
|
+
cacheWriteUsdPerMillion: z.number().nonnegative().optional(),
|
|
2504
|
+
outputUsdPerMillion: z.number().nonnegative()
|
|
2505
|
+
}).strict().optional(),
|
|
2506
|
+
actualCostUsd: z.number().nonnegative().optional(),
|
|
2507
|
+
estimatedCostUsd: z.number().nonnegative().optional(),
|
|
2508
|
+
costUnknown: z.boolean().optional(),
|
|
2509
|
+
usageUnknown: z.boolean().optional()
|
|
2510
|
+
}).strict();
|
|
2511
|
+
const BenchmarkErrorSchema = z.object({
|
|
2512
|
+
class: z.string().min(1),
|
|
2513
|
+
message: z.string(),
|
|
2514
|
+
code: z.string().min(1).optional(),
|
|
2515
|
+
status: z.number().int().min(100).max(599).optional()
|
|
2516
|
+
}).strict();
|
|
2517
|
+
const ResponseMetadataSchema = z.object({
|
|
2518
|
+
providerModel: z.string().min(1),
|
|
2519
|
+
providerDurationMs: z.number().nonnegative(),
|
|
2520
|
+
finishReason: z.string().nullable(),
|
|
2521
|
+
producedAt: z.string().datetime()
|
|
2522
|
+
}).strict();
|
|
2523
|
+
const CacheIdentityShape = {
|
|
2524
|
+
kind: z.literal("agent-eval/public-benchmark-model-response"),
|
|
2525
|
+
callId: z.string().min(1),
|
|
2526
|
+
runIdentitySha256: z.string().regex(SHA256),
|
|
2527
|
+
caseId: z.string().min(1),
|
|
2528
|
+
repetition: z.number().int().nonnegative()
|
|
2529
|
+
};
|
|
2530
|
+
const SuccessCacheEntryWithoutDigestSchema = z.object({
|
|
2531
|
+
...CacheIdentityShape,
|
|
2532
|
+
status: z.literal("succeeded"),
|
|
2533
|
+
response: z.json(),
|
|
2534
|
+
metadata: ResponseMetadataSchema,
|
|
2535
|
+
receipt: CostReceiptInputSchema
|
|
2536
|
+
}).strict();
|
|
2537
|
+
const FailureCacheEntryWithoutDigestSchema = z.object({
|
|
2538
|
+
...CacheIdentityShape,
|
|
2539
|
+
status: z.literal("failed"),
|
|
2540
|
+
error: BenchmarkErrorSchema,
|
|
2541
|
+
receipt: CostReceiptInputSchema
|
|
2542
|
+
}).strict();
|
|
2543
|
+
const CacheEntryWithoutDigestSchema = z.discriminatedUnion("status", [SuccessCacheEntryWithoutDigestSchema, FailureCacheEntryWithoutDigestSchema]);
|
|
2544
|
+
const CacheEntrySchema = z.discriminatedUnion("status", [SuccessCacheEntryWithoutDigestSchema.extend({ entrySha256: z.string().regex(SHA256) }), FailureCacheEntryWithoutDigestSchema.extend({ entrySha256: z.string().regex(SHA256) })]);
|
|
2545
|
+
function publicBenchmarkCallId(identity) {
|
|
2546
|
+
assertCacheIdentity(identity);
|
|
2547
|
+
return `analyst-benchmark-${hashCanonical({
|
|
2548
|
+
runIdentitySha256: identity.runIdentitySha256,
|
|
2549
|
+
caseId: identity.caseId,
|
|
2550
|
+
repetition: identity.repetition
|
|
2551
|
+
}).slice(7)}`;
|
|
2552
|
+
}
|
|
2553
|
+
function readPublicBenchmarkResponseCache(cacheDirectory, identity) {
|
|
2554
|
+
const callId = publicBenchmarkCallId(identity);
|
|
2555
|
+
const path = responseCachePath(cacheDirectory, callId);
|
|
2556
|
+
if (!existsSync(path)) return void 0;
|
|
2557
|
+
const metadata = lstatSync(path);
|
|
2558
|
+
if (!metadata.isFile() || metadata.isSymbolicLink()) throw new ValidationError(`benchmark response cache must be a real file: ${path}`);
|
|
2559
|
+
let parsed;
|
|
2560
|
+
try {
|
|
2561
|
+
parsed = JSON.parse(readFileSync(path, "utf8"));
|
|
2562
|
+
} catch (error) {
|
|
2563
|
+
throw new ValidationError(`benchmark response cache contains invalid JSON: ${path}`, { cause: error });
|
|
2564
|
+
}
|
|
2565
|
+
const entry = parseCacheEntry(parsed, path);
|
|
2566
|
+
if (entry.callId !== callId || entry.runIdentitySha256 !== identity.runIdentitySha256 || entry.caseId !== identity.caseId || entry.repetition !== identity.repetition) throw new ValidationError(`benchmark response cache identity does not match: ${path}`);
|
|
2567
|
+
return entry;
|
|
2568
|
+
}
|
|
2569
|
+
function writePublicBenchmarkResponseCache(cacheDirectory, entry) {
|
|
2570
|
+
const expectedCallId = publicBenchmarkCallId(entry);
|
|
2571
|
+
if (entry.callId !== expectedCallId) throw new ValidationError("benchmark response cache callId does not match its identity");
|
|
2572
|
+
const validated = JSON.parse(JSON.stringify(CacheEntryWithoutDigestSchema.parse(entry)));
|
|
2573
|
+
const complete = {
|
|
2574
|
+
...validated,
|
|
2575
|
+
entrySha256: hashCanonical(validated).slice(7)
|
|
2576
|
+
};
|
|
2577
|
+
const path = responseCachePath(cacheDirectory, complete.callId);
|
|
2578
|
+
const content = `${canonicalString(complete)}\n`;
|
|
2579
|
+
withLedgerFileLock(path, fileContext(), () => {
|
|
2580
|
+
if (existsSync(path)) {
|
|
2581
|
+
const existing = readPublicBenchmarkResponseCache(cacheDirectory, complete);
|
|
2582
|
+
if (!existing || canonicalString(existing) !== canonicalString(complete)) throw new ValidationError(`benchmark response cache conflicts with existing file: ${path}`);
|
|
2583
|
+
return;
|
|
2584
|
+
}
|
|
2585
|
+
writeLedgerFileAtomically(path, content, fileContext());
|
|
2586
|
+
});
|
|
2587
|
+
return complete;
|
|
2588
|
+
}
|
|
2589
|
+
function parseCacheEntry(value, path) {
|
|
2590
|
+
let parsed;
|
|
2591
|
+
try {
|
|
2592
|
+
parsed = CacheEntrySchema.parse(value);
|
|
2593
|
+
} catch (error) {
|
|
2594
|
+
throw new ValidationError(`benchmark response cache has an invalid shape: ${path}`, { cause: error });
|
|
2595
|
+
}
|
|
2596
|
+
const { entrySha256, ...withoutDigest } = parsed;
|
|
2597
|
+
if (entrySha256 !== hashCanonical(withoutDigest).slice(7)) throw new ValidationError(`benchmark response cache digest does not match: ${path}`);
|
|
2598
|
+
return parsed;
|
|
2599
|
+
}
|
|
2600
|
+
function responseCachePath(cacheDirectory, callId) {
|
|
2601
|
+
const directory = nodePath.resolve(cacheDirectory);
|
|
2602
|
+
const path = nodePath.resolve(directory, `${hashCanonical(callId).slice(7)}.json`);
|
|
2603
|
+
if (!isPathInsideDirectory(directory, path, nodePath)) throw new ValidationError("benchmark response cache path escapes its directory");
|
|
2604
|
+
return path;
|
|
2605
|
+
}
|
|
2606
|
+
function isPathInsideDirectory(directory, candidate, pathOperations = nodePath) {
|
|
2607
|
+
const relative = pathOperations.relative(directory, candidate);
|
|
2608
|
+
return relative !== "" && relative !== ".." && !relative.startsWith(`..${pathOperations.sep}`) && !pathOperations.isAbsolute(relative);
|
|
2609
|
+
}
|
|
2610
|
+
function assertCacheIdentity(identity) {
|
|
2611
|
+
if (!SHA256.test(identity.runIdentitySha256)) throw new ValidationError("benchmark response cache requires a SHA-256 run identity");
|
|
2612
|
+
if (!identity.caseId.trim()) throw new ValidationError("benchmark response cache requires a case id");
|
|
2613
|
+
if (!Number.isSafeInteger(identity.repetition) || identity.repetition < 0) throw new ValidationError("benchmark response cache repetition must be non-negative");
|
|
2614
|
+
}
|
|
2615
|
+
function fileContext() {
|
|
2616
|
+
return {
|
|
2617
|
+
subject: "benchmark response cache",
|
|
2618
|
+
integrityError: (message, options) => new ValidationError(message, options)
|
|
2619
|
+
};
|
|
2620
|
+
}
|
|
2621
|
+
//#endregion
|
|
2622
|
+
//#region src/analyst/benchmark-public-model.ts
|
|
2623
|
+
const TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS = [
|
|
2624
|
+
4096,
|
|
2625
|
+
2048,
|
|
2626
|
+
1024,
|
|
2627
|
+
512,
|
|
2628
|
+
256,
|
|
2629
|
+
128,
|
|
2630
|
+
64
|
|
2631
|
+
];
|
|
2632
|
+
function createPublicBenchmarkModelRunner(dataset, config) {
|
|
2633
|
+
const model = requiredString(config.model, "model");
|
|
2634
|
+
const baseUrl = requiredString(config.baseUrl, "baseUrl");
|
|
2635
|
+
const apiKey = requiredString(config.apiKey, "apiKey");
|
|
2636
|
+
const maxOutputTokens = positiveSafeInteger(config.maxOutputTokens, "maxOutputTokens");
|
|
2637
|
+
const timeoutMs = positiveSafeInteger(config.timeoutMs, "timeoutMs");
|
|
2638
|
+
const costLedger = config.costLedger ?? new CostLedger();
|
|
2639
|
+
const durability = config.durability ? {
|
|
2640
|
+
runIdentitySha256: requiredString(config.durability.runIdentitySha256, "durability.runIdentitySha256"),
|
|
2641
|
+
responseCacheDir: requiredString(config.durability.responseCacheDir, "durability.responseCacheDir")
|
|
2642
|
+
} : void 0;
|
|
2643
|
+
const actor = dataset === "agentrx" ? "agentrx-root-cause-localizer" : "codetracebench-step-localizer";
|
|
2644
|
+
const outputAdapter = dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-step";
|
|
2645
|
+
const llmOptions = {
|
|
2646
|
+
baseUrl,
|
|
2647
|
+
apiKey,
|
|
2648
|
+
maximumAttempts: 1,
|
|
2649
|
+
jsonSchemaTransport: "json-object",
|
|
2650
|
+
jsonPayloadMode: "exact",
|
|
2651
|
+
thinking: "disabled",
|
|
2652
|
+
...config.fetchImpl ? { fetch: config.fetchImpl } : {}
|
|
2653
|
+
};
|
|
2654
|
+
return {
|
|
2655
|
+
id: "model",
|
|
2656
|
+
async analyze(input, context) {
|
|
2657
|
+
const trajectoryId = trajectoryIdFromCaseId(dataset, context.caseId);
|
|
2658
|
+
const costTags = {
|
|
2659
|
+
analystId: actor,
|
|
2660
|
+
benchmarkCaseId: context.caseId,
|
|
2661
|
+
benchmarkRepetition: String(context.repetition)
|
|
2662
|
+
};
|
|
2663
|
+
let rawPredictions = [];
|
|
2664
|
+
let modelFindings = [];
|
|
2665
|
+
let providerModel = model;
|
|
2666
|
+
let producedAt;
|
|
2667
|
+
let modelMetadata = {
|
|
2668
|
+
analysisMode: "single-pass",
|
|
2669
|
+
outputAdapter,
|
|
2670
|
+
protocolSha256: publicBenchmarkProtocolSha256(dataset)
|
|
2671
|
+
};
|
|
2672
|
+
try {
|
|
2673
|
+
if (!input.traceStore) throw new Error(`${dataset} model runner requires a trace store`);
|
|
2674
|
+
const preparedContext = await prepareSingleTraceContext(input.traceStore, context);
|
|
2675
|
+
if (preparedContext === void 0) throw new Error(`${dataset} trace '${trajectoryId}' has no readable spans`);
|
|
2676
|
+
const request = {
|
|
2677
|
+
model,
|
|
2678
|
+
messages: [{
|
|
2679
|
+
role: "system",
|
|
2680
|
+
content: publicBenchmarkSystemPrompt(dataset)
|
|
2681
|
+
}, {
|
|
2682
|
+
role: "user",
|
|
2683
|
+
content: `TRACE DATA:\n${preparedContext}\n\nReturn the analysis JSON object.`
|
|
2684
|
+
}],
|
|
2685
|
+
jsonMode: true,
|
|
2686
|
+
thinking: "disabled",
|
|
2687
|
+
maxTokens: maxOutputTokens,
|
|
2688
|
+
timeoutMs
|
|
2689
|
+
};
|
|
2690
|
+
const cacheIdentity = durability ? {
|
|
2691
|
+
runIdentitySha256: durability.runIdentitySha256,
|
|
2692
|
+
caseId: context.caseId,
|
|
2693
|
+
repetition: context.repetition
|
|
2694
|
+
} : void 0;
|
|
2695
|
+
const callId = cacheIdentity ? publicBenchmarkCallId(cacheIdentity) : void 0;
|
|
2696
|
+
const cached = cacheIdentity ? readPublicBenchmarkResponseCache(durability.responseCacheDir, cacheIdentity) : void 0;
|
|
2697
|
+
if (cached) {
|
|
2698
|
+
const receipt = settleCachedResponse(costLedger, cached);
|
|
2699
|
+
modelMetadata = {
|
|
2700
|
+
...modelMetadata,
|
|
2701
|
+
responseSource: "durable-cache",
|
|
2702
|
+
cost: costReceiptMetadata(receipt)
|
|
2703
|
+
};
|
|
2704
|
+
if (cached.status === "failed") return {
|
|
2705
|
+
findings: [],
|
|
2706
|
+
usage: usageReceiptFromCostLedger(costLedger, {
|
|
2707
|
+
channel: "analyst",
|
|
2708
|
+
tags: costTags
|
|
2709
|
+
}),
|
|
2710
|
+
error: cached.error,
|
|
2711
|
+
metadata: modelMetadata
|
|
2712
|
+
};
|
|
2713
|
+
const response = parsePublicBenchmarkModelResponse(dataset, cached.response);
|
|
2714
|
+
rawPredictions = response.findings;
|
|
2715
|
+
providerModel = cached.metadata.providerModel;
|
|
2716
|
+
producedAt = cached.metadata.producedAt;
|
|
2717
|
+
modelMetadata = {
|
|
2718
|
+
...modelMetadata,
|
|
2719
|
+
report: response.report,
|
|
2720
|
+
providerModel: cached.metadata.providerModel,
|
|
2721
|
+
providerDurationMs: cached.metadata.providerDurationMs,
|
|
2722
|
+
finishReason: cached.metadata.finishReason
|
|
2723
|
+
};
|
|
2724
|
+
} else {
|
|
2725
|
+
assertNoSettledResponseWithoutCache(costLedger, callId);
|
|
2726
|
+
let completedResult;
|
|
2727
|
+
const paid = await costLedger.runPaidCall({
|
|
2728
|
+
...callId ? { callId } : {},
|
|
2729
|
+
channel: "analyst",
|
|
2730
|
+
phase: "analyst.public-benchmark",
|
|
2731
|
+
actor,
|
|
2732
|
+
model,
|
|
2733
|
+
signal: context.signal,
|
|
2734
|
+
maximumCharge: maximumChargeForLlmRequest(request, llmOptions),
|
|
2735
|
+
tags: costTags,
|
|
2736
|
+
execute: async (signal, providerCallId) => {
|
|
2737
|
+
try {
|
|
2738
|
+
const completed = await callLlmJson(request, {
|
|
2739
|
+
...llmOptions,
|
|
2740
|
+
signal,
|
|
2741
|
+
idempotencyKey: providerCallId
|
|
2742
|
+
});
|
|
2743
|
+
completedResult = completed.result;
|
|
2744
|
+
const response = parsePublicBenchmarkModelResponse(dataset, completed.value);
|
|
2745
|
+
const responseProducedAt = (/* @__PURE__ */ new Date()).toISOString();
|
|
2746
|
+
if (cacheIdentity) writePublicBenchmarkResponseCache(durability.responseCacheDir, {
|
|
2747
|
+
kind: "agent-eval/public-benchmark-model-response",
|
|
2748
|
+
...cacheIdentity,
|
|
2749
|
+
callId: providerCallId,
|
|
2750
|
+
status: "succeeded",
|
|
2751
|
+
response,
|
|
2752
|
+
metadata: {
|
|
2753
|
+
providerModel: completed.result.model,
|
|
2754
|
+
providerDurationMs: completed.result.durationMs,
|
|
2755
|
+
finishReason: completed.result.finishReason ?? null,
|
|
2756
|
+
producedAt: responseProducedAt
|
|
2757
|
+
},
|
|
2758
|
+
receipt: costReceiptFromLlm(completed.result)
|
|
2759
|
+
});
|
|
2760
|
+
return {
|
|
2761
|
+
...completed,
|
|
2762
|
+
response,
|
|
2763
|
+
producedAt: responseProducedAt
|
|
2764
|
+
};
|
|
2765
|
+
} catch (error) {
|
|
2766
|
+
if (cacheIdentity) writePublicBenchmarkResponseCache(durability.responseCacheDir, {
|
|
2767
|
+
kind: "agent-eval/public-benchmark-model-response",
|
|
2768
|
+
...cacheIdentity,
|
|
2769
|
+
callId: providerCallId,
|
|
2770
|
+
status: "failed",
|
|
2771
|
+
error: publicBenchmarkError(error, [apiKey]),
|
|
2772
|
+
receipt: receiptForProviderFailure(error, completedResult, model)
|
|
2773
|
+
});
|
|
2774
|
+
throw error;
|
|
2775
|
+
}
|
|
2776
|
+
},
|
|
2777
|
+
receipt: ({ result }) => costReceiptFromLlm(result),
|
|
2778
|
+
receiptFromError: (error) => receiptForProviderFailure(error, completedResult, model)
|
|
2779
|
+
});
|
|
2780
|
+
if (!paid.succeeded) throw paid.error;
|
|
2781
|
+
const response = paid.value.response;
|
|
2782
|
+
rawPredictions = response.findings;
|
|
2783
|
+
providerModel = paid.value.result.model;
|
|
2784
|
+
producedAt = paid.value.producedAt;
|
|
2785
|
+
modelMetadata = {
|
|
2786
|
+
...modelMetadata,
|
|
2787
|
+
responseSource: "provider",
|
|
2788
|
+
report: response.report,
|
|
2789
|
+
providerModel: paid.value.result.model,
|
|
2790
|
+
providerDurationMs: paid.value.result.durationMs,
|
|
2791
|
+
finishReason: paid.value.result.finishReason ?? null,
|
|
2792
|
+
cost: costReceiptMetadata(paid.receipt)
|
|
2793
|
+
};
|
|
2794
|
+
}
|
|
2795
|
+
modelFindings = await publicBenchmarkPredictionsToFindings({
|
|
2796
|
+
dataset,
|
|
2797
|
+
trajectoryId,
|
|
2798
|
+
predictions: rawPredictions,
|
|
2799
|
+
store: input.traceStore,
|
|
2800
|
+
analystId: "model",
|
|
2801
|
+
providerModel,
|
|
2802
|
+
producedAt: requiredString(producedAt ?? "", "finding producedAt"),
|
|
2803
|
+
...context.signal ? { signal: context.signal } : {}
|
|
2804
|
+
});
|
|
2805
|
+
if (dataset === "codetracebench") await validateCodeTraceFindingEvidence({
|
|
2806
|
+
trajectoryId,
|
|
2807
|
+
findings: modelFindings,
|
|
2808
|
+
store: input.traceStore,
|
|
2809
|
+
...context.signal ? { signal: context.signal } : {}
|
|
2810
|
+
});
|
|
2811
|
+
return {
|
|
2812
|
+
findings: modelFindings,
|
|
2813
|
+
usage: usageReceiptFromCostLedger(costLedger, {
|
|
2814
|
+
channel: "analyst",
|
|
2815
|
+
tags: costTags
|
|
2816
|
+
}),
|
|
2817
|
+
metadata: modelMetadata
|
|
2818
|
+
};
|
|
2819
|
+
} catch (error) {
|
|
2820
|
+
if (context.signal?.aborted) throw error;
|
|
2821
|
+
if (isPaidCallControlError(error)) throw error;
|
|
2822
|
+
return {
|
|
2823
|
+
findings: [],
|
|
2824
|
+
usage: usageReceiptFromCostLedger(costLedger, {
|
|
2825
|
+
channel: "analyst",
|
|
2826
|
+
tags: costTags
|
|
2827
|
+
}),
|
|
2828
|
+
error: publicBenchmarkError(error, [apiKey]),
|
|
2829
|
+
metadata: {
|
|
2830
|
+
...modelMetadata,
|
|
2831
|
+
rawPredictions,
|
|
2832
|
+
acceptedFindings: modelFindings
|
|
2833
|
+
}
|
|
2834
|
+
};
|
|
2835
|
+
}
|
|
2836
|
+
}
|
|
2837
|
+
};
|
|
2838
|
+
}
|
|
2839
|
+
function settleCachedResponse(costLedger, cached) {
|
|
2840
|
+
const settled = costLedger.list().find((receipt) => receipt.callId === cached.callId);
|
|
2841
|
+
const pending = costLedger.listPending?.().find((record) => record.callId === cached.callId);
|
|
2842
|
+
if (settled && pending) throw new CostCallConflictError(`benchmark response '${cached.callId}' is both pending and settled`, { callId: cached.callId });
|
|
2843
|
+
const receipt = pending ? costLedger.reconcile(cached.callId, cached.receipt, { ...cached.status === "failed" ? { failed: true } : {} }) : settled;
|
|
2844
|
+
if (!receipt) throw new CostCallConflictError(`benchmark response cache '${cached.callId}' has no matching cost record`, { callId: cached.callId });
|
|
2845
|
+
assertCacheReceiptMatches(cached, receipt);
|
|
2846
|
+
return receipt;
|
|
2847
|
+
}
|
|
2848
|
+
function assertNoSettledResponseWithoutCache(costLedger, callId) {
|
|
2849
|
+
if (!callId) return;
|
|
2850
|
+
if (costLedger.list().some((receipt) => receipt.callId === callId)) throw new CostCallConflictError(`settled benchmark call '${callId}' has no durable response cache`, { callId });
|
|
2851
|
+
}
|
|
2852
|
+
function assertCacheReceiptMatches(cached, receipt) {
|
|
2853
|
+
const expected = cached.receipt;
|
|
2854
|
+
if (receipt.callId !== cached.callId || receipt.model !== expected.model || receipt.inputTokens !== expected.inputTokens || receipt.outputTokens !== expected.outputTokens || (receipt.reasoningTokens ?? 0) !== (expected.reasoningTokens ?? 0) || (receipt.cachedTokens ?? 0) !== (expected.cachedTokens ?? 0) || (receipt.cacheWriteTokens ?? 0) !== (expected.cacheWriteTokens ?? 0) || expected.actualCostUsd !== void 0 && receipt.actualCostUsd !== expected.actualCostUsd || expected.estimatedCostUsd !== void 0 && receipt.estimatedCostUsd !== expected.estimatedCostUsd || expected.costUnknown === true && !receipt.costUnknown || expected.usageUnknown === true && !receipt.usageUnknown || cached.status === "succeeded" && receipt.error !== void 0 || cached.status === "failed" && receipt.error === void 0) throw new CostCallConflictError(`benchmark response cache receipt does not match cost record '${cached.callId}'`, {
|
|
2855
|
+
callId: cached.callId,
|
|
2856
|
+
receipt
|
|
2857
|
+
});
|
|
2858
|
+
}
|
|
2859
|
+
function isPaidCallControlError(error) {
|
|
2860
|
+
return error instanceof CostAccountingIncompleteError || error instanceof CostCallConflictError || error instanceof CostCeilingReachedError || error instanceof CostLedgerPersistenceError || error instanceof CostReceiptCaptureError || error instanceof CostReservationExceededError;
|
|
2861
|
+
}
|
|
2862
|
+
function receiptForProviderFailure(error, completedResult, model) {
|
|
2863
|
+
if (completedResult) return costReceiptFromLlm(completedResult);
|
|
2864
|
+
if (error instanceof Error) {
|
|
2865
|
+
const captured = costReceiptFromLlmError(error);
|
|
2866
|
+
if (captured) return captured;
|
|
2867
|
+
}
|
|
2868
|
+
return {
|
|
2869
|
+
model,
|
|
2870
|
+
inputTokens: 0,
|
|
2871
|
+
outputTokens: 0,
|
|
2872
|
+
costUnknown: true,
|
|
2873
|
+
usageUnknown: true
|
|
2874
|
+
};
|
|
2875
|
+
}
|
|
2876
|
+
function costReceiptMetadata(receipt) {
|
|
2877
|
+
if (receipt.actualCostUsd !== void 0) return {
|
|
2878
|
+
source: "provider",
|
|
2879
|
+
actualCostUsd: receipt.actualCostUsd
|
|
2880
|
+
};
|
|
2881
|
+
if (receipt.estimatedCostUsd !== void 0) return {
|
|
2882
|
+
source: "external-estimate",
|
|
2883
|
+
estimatedCostUsd: receipt.estimatedCostUsd
|
|
2884
|
+
};
|
|
2885
|
+
if (receipt.pricing) return {
|
|
2886
|
+
source: "agent-eval-model-pricing",
|
|
2887
|
+
estimatedCostUsd: receipt.costUsd,
|
|
2888
|
+
ratesPerThousandTokens: receipt.pricing
|
|
2889
|
+
};
|
|
2890
|
+
return {
|
|
2891
|
+
source: "unknown",
|
|
2892
|
+
estimatedCostUsd: null
|
|
2893
|
+
};
|
|
2894
|
+
}
|
|
2895
|
+
const ModelSeveritySchema = z.enum([
|
|
2896
|
+
"critical",
|
|
2897
|
+
"high",
|
|
2898
|
+
"medium",
|
|
2899
|
+
"low",
|
|
2900
|
+
"info"
|
|
2901
|
+
]);
|
|
2902
|
+
const PublicBenchmarkPredictionSchema = z.object({
|
|
2903
|
+
step: z.number().int().positive(),
|
|
2904
|
+
severity: ModelSeveritySchema,
|
|
2905
|
+
claim: z.string().min(1),
|
|
2906
|
+
confidence: z.number().min(0).max(1),
|
|
2907
|
+
rationale: z.string().min(1).optional(),
|
|
2908
|
+
recommended_action: z.string().min(1).optional()
|
|
2909
|
+
}).strict();
|
|
2910
|
+
const AgentRxCategorySchema = z.enum([
|
|
2911
|
+
"instruction-plan-adherence-failure",
|
|
2912
|
+
"invention-of-new-information",
|
|
2913
|
+
"invalid-invocation",
|
|
2914
|
+
"misinterpretation-of-tool-output-handoff-failure",
|
|
2915
|
+
"intent-plan-misalignment",
|
|
2916
|
+
"underspecified-user-intent",
|
|
2917
|
+
"intent-not-supported",
|
|
2918
|
+
"guardrails-triggered",
|
|
2919
|
+
"system-failure",
|
|
2920
|
+
"inconclusive"
|
|
2921
|
+
]);
|
|
2922
|
+
const CodeTraceModelResponseSchema = z.object({
|
|
2923
|
+
report: z.string().min(1).max(4e3),
|
|
2924
|
+
findings: z.array(PublicBenchmarkPredictionSchema).max(200)
|
|
2925
|
+
}).strict();
|
|
2926
|
+
const AgentRxModelResponseSchema = z.object({
|
|
2927
|
+
report: z.string().min(1).max(4e3),
|
|
2928
|
+
findings: z.array(PublicBenchmarkPredictionSchema.extend({ category: AgentRxCategorySchema }).strict()).max(1)
|
|
2929
|
+
}).strict();
|
|
2930
|
+
function parsePublicBenchmarkModelResponse(dataset, value) {
|
|
2931
|
+
return dataset === "agentrx" ? AgentRxModelResponseSchema.parse(value) : CodeTraceModelResponseSchema.parse(value);
|
|
2932
|
+
}
|
|
2933
|
+
async function publicBenchmarkPredictionsToFindings(options) {
|
|
2934
|
+
if (options.predictions.length === 0) return [];
|
|
2935
|
+
const evidenceByStep = await resolveAssistantStepEvidence({
|
|
2936
|
+
trajectoryId: options.trajectoryId,
|
|
2937
|
+
steps: options.predictions.map((prediction) => prediction.step),
|
|
2938
|
+
store: options.store,
|
|
2939
|
+
...options.signal ? { signal: options.signal } : {}
|
|
2940
|
+
});
|
|
2941
|
+
if (options.dataset === "agentrx") {
|
|
2942
|
+
const prediction = options.predictions[0];
|
|
2943
|
+
if (!prediction.category) throw new Error("AgentRx model output is missing its failure category");
|
|
2944
|
+
const [finding] = agentRxPredictionsToFindings(options.trajectoryId, [{
|
|
2945
|
+
failure_case: prediction.category,
|
|
2946
|
+
step_number: prediction.step,
|
|
2947
|
+
description: prediction.rationale ?? prediction.claim
|
|
2948
|
+
}], {
|
|
2949
|
+
analystId: options.analystId,
|
|
2950
|
+
producedAt: options.producedAt,
|
|
2951
|
+
confidence: prediction.confidence
|
|
2952
|
+
});
|
|
2953
|
+
if (!finding) throw new Error("AgentRx output adapter produced no root-cause finding");
|
|
2954
|
+
return [{
|
|
2955
|
+
...finding,
|
|
2956
|
+
evidence_refs: [evidenceByStep.get(prediction.step)],
|
|
2957
|
+
metadata: {
|
|
2958
|
+
...finding.metadata,
|
|
2959
|
+
model: options.providerModel
|
|
2960
|
+
}
|
|
2961
|
+
}];
|
|
2962
|
+
}
|
|
2963
|
+
const byStep = /* @__PURE__ */ new Map();
|
|
2964
|
+
for (const prediction of options.predictions) if (!byStep.has(prediction.step)) byStep.set(prediction.step, prediction);
|
|
2965
|
+
return [...byStep].sort(([left], [right]) => left - right).map(([step, prediction]) => makeFinding({
|
|
2966
|
+
analyst_id: options.analystId,
|
|
2967
|
+
area: "incorrect",
|
|
2968
|
+
subject: `incorrect-step-${step}`,
|
|
2969
|
+
claim: `Step ${step} is incorrect. ${prediction.claim}`,
|
|
2970
|
+
rationale: prediction.rationale,
|
|
2971
|
+
severity: prediction.severity,
|
|
2972
|
+
confidence: prediction.confidence,
|
|
2973
|
+
evidence_refs: [evidenceByStep.get(step)],
|
|
2974
|
+
recommended_action: prediction.recommended_action,
|
|
2975
|
+
metadata: {
|
|
2976
|
+
analysis_mode: "single-pass",
|
|
2977
|
+
model: options.providerModel
|
|
2978
|
+
},
|
|
2979
|
+
produced_at: options.producedAt,
|
|
2980
|
+
id_basis: `incorrect-step-${step}`
|
|
2981
|
+
}));
|
|
2982
|
+
}
|
|
2983
|
+
function publicBenchmarkSystemPrompt(dataset) {
|
|
2984
|
+
return `${dataset === "agentrx" ? AGENT_RX_PROMPT : CODE_TRACE_BENCH_ANALYST_PROMPT}
|
|
2985
|
+
|
|
2986
|
+
Each finding must contain only:
|
|
2987
|
+
- "step": a positive integer matching an existing assistant LLM span named step-<n>
|
|
2988
|
+
- "severity": "critical", "high", "medium", "low", or "info"
|
|
2989
|
+
- "claim": one sentence
|
|
2990
|
+
- "confidence": a number from 0 through 1
|
|
2991
|
+
- optional "rationale" and "recommended_action" strings
|
|
2992
|
+
${dataset === "agentrx" ? `- "category": one allowed failure category listed above` : ""}
|
|
2993
|
+
|
|
2994
|
+
Return exactly one JSON object with:
|
|
2995
|
+
- "report": a concise evidence-based explanation, at most 4000 characters
|
|
2996
|
+
- "findings": the strict finding array
|
|
2997
|
+
Use an empty findings array when the trace does not support a finding.
|
|
2998
|
+
Do not return a bare array, markdown, trace URIs, copied excerpts, or fields not listed above.
|
|
2999
|
+
The runner constructs exact trace URIs and action previews from each selected step.`;
|
|
3000
|
+
}
|
|
3001
|
+
function publicBenchmarkProtocolSha256(dataset) {
|
|
3002
|
+
return sha256Digest(JSON.stringify({
|
|
3003
|
+
dataset,
|
|
3004
|
+
systemPrompt: publicBenchmarkSystemPrompt(dataset),
|
|
3005
|
+
transport: {
|
|
3006
|
+
attempts: 1,
|
|
3007
|
+
jsonMode: true,
|
|
3008
|
+
thinking: "disabled"
|
|
3009
|
+
},
|
|
3010
|
+
traceProjectionAttributeByteCaps: TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS,
|
|
3011
|
+
evidence: {
|
|
3012
|
+
location: "model-selected-positive-integer-assistant-step",
|
|
3013
|
+
uri: "deterministic-trace-uri",
|
|
3014
|
+
excerpt: `exact-action-prefix-512`
|
|
3015
|
+
}
|
|
3016
|
+
}));
|
|
3017
|
+
}
|
|
3018
|
+
async function prepareSingleTraceContext(store, context) {
|
|
3019
|
+
const storeContext = context.signal ? { signal: context.signal } : void 0;
|
|
3020
|
+
const overview = await store.getOverview(void 0, storeContext);
|
|
3021
|
+
if (overview.total_traces !== 1 || overview.sample_trace_ids.length !== 1) throw new Error(`public model benchmark requires exactly one trace, received ${overview.total_traces}`);
|
|
3022
|
+
const traceId = overview.sample_trace_ids[0];
|
|
3023
|
+
for (const perAttributeByteCap of TRACE_PROJECTION_ATTRIBUTE_BYTE_CAPS) {
|
|
3024
|
+
const viewed = await store.viewTrace({
|
|
3025
|
+
trace_id: traceId,
|
|
3026
|
+
per_attribute_byte_cap: perAttributeByteCap
|
|
3027
|
+
}, storeContext);
|
|
3028
|
+
if (!viewed.spans) continue;
|
|
3029
|
+
return JSON.stringify({
|
|
3030
|
+
trace_id: traceId,
|
|
3031
|
+
per_attribute_byte_cap: perAttributeByteCap,
|
|
3032
|
+
spans: viewed.spans
|
|
3033
|
+
});
|
|
3034
|
+
}
|
|
3035
|
+
}
|
|
3036
|
+
const AGENT_RX_PROMPT = `Analyze exactly one failed agent trajectory.
|
|
3037
|
+
Find the first unrecoverable critical failure, not every later symptom.
|
|
3038
|
+
Inspect the complete supplied trace data.
|
|
3039
|
+
Emit zero findings only when the trace does not contain enough evidence.
|
|
3040
|
+
Otherwise emit exactly one finding.
|
|
3041
|
+
Its category MUST be exactly one of:
|
|
3042
|
+
instruction-plan-adherence-failure
|
|
3043
|
+
invention-of-new-information
|
|
3044
|
+
invalid-invocation
|
|
3045
|
+
misinterpretation-of-tool-output-handoff-failure
|
|
3046
|
+
intent-plan-misalignment
|
|
3047
|
+
underspecified-user-intent
|
|
3048
|
+
intent-not-supported
|
|
3049
|
+
guardrails-triggered
|
|
3050
|
+
system-failure
|
|
3051
|
+
inconclusive
|
|
3052
|
+
Its step is the positive integer n from the first unrecoverable assistant span named step-<n>.`;
|
|
3053
|
+
const CODE_TRACE_BENCH_ANALYST_PROMPT = `Analyze exactly one coding-agent trajectory and its attached final verification.
|
|
3054
|
+
Your task is the CodeTraceBench incorrect-step task: identify every wrong state-changing action, bad hypothesis that drives an action, and regression.
|
|
3055
|
+
An incorrect step remains incorrect when the agent later recovers or the final verification passes.
|
|
3056
|
+
Inspect the complete supplied trace data.
|
|
3057
|
+
Use the final-verification outcome as evidence about the final state, not as a rule for whether earlier steps were incorrect.
|
|
3058
|
+
For each candidate, inspect the assistant action and its following observation.
|
|
3059
|
+
Label a failed command when the assistant caused it through a wrong action or unsupported hypothesis.
|
|
3060
|
+
Label the later corrective action only when that action is itself wrong.
|
|
3061
|
+
Do not label a diagnostic probe merely because it exposes an earlier defect.
|
|
3062
|
+
Do not label a redundant but correct read or search; CodeTraceBench scores unuseful steps separately, and this run scores incorrect steps only.
|
|
3063
|
+
Do not label a step solely because final verification failed.
|
|
3064
|
+
When final verification is unavailable, use only directly observed trajectory evidence.
|
|
3065
|
+
Emit one finding per incorrect assistant step.
|
|
3066
|
+
Each finding's step MUST be the positive integer n from an existing assistant LLM span named step-<n>.
|
|
3067
|
+
Never select an EVALUATOR, TOOL, CHAIN, final-verification, benchmark-verification, or message-<n> span.
|
|
3068
|
+
Before emitting a finding, inspect its candidate span's attributes.content and describe only the action shown there.
|
|
3069
|
+
When the trajectory has no incorrect steps, return an empty findings array.`;
|
|
3070
|
+
function trajectoryIdFromCaseId(dataset, caseId) {
|
|
3071
|
+
const prefix = dataset === "agentrx" ? "agentrx:" : "codetrace:";
|
|
3072
|
+
if (!caseId.startsWith(prefix) || caseId.length === prefix.length) throw new Error(`unexpected ${dataset} benchmark case id '${caseId}'`);
|
|
3073
|
+
return caseId.slice(prefix.length);
|
|
3074
|
+
}
|
|
3075
|
+
//#endregion
|
|
3076
|
+
//#region src/analyst/benchmark-command-persistence.ts
|
|
3077
|
+
const ANALYST_BENCHMARK_INITIALIZATION_COMPLETE_FILE = "initialization-complete.json";
|
|
3078
|
+
async function openOutputDirectory(outDir, resume) {
|
|
3079
|
+
const directory = resolve(outDir);
|
|
3080
|
+
if (resume) {
|
|
3081
|
+
let outputStat;
|
|
3082
|
+
try {
|
|
3083
|
+
outputStat = await lstat(directory);
|
|
3084
|
+
} catch (error) {
|
|
3085
|
+
if (isNodeError(error, "ENOENT")) throw new Error(`cannot resume missing benchmark output directory: ${directory}`);
|
|
3086
|
+
throw error;
|
|
3087
|
+
}
|
|
3088
|
+
if (!outputStat.isDirectory() || outputStat.isSymbolicLink()) throw new Error(`benchmark output must be a real directory: ${directory}`);
|
|
3089
|
+
} else {
|
|
3090
|
+
await mkdir(dirname(directory), { recursive: true });
|
|
3091
|
+
try {
|
|
3092
|
+
await mkdir(directory);
|
|
3093
|
+
} catch (error) {
|
|
3094
|
+
if (isNodeError(error, "EEXIST")) throw new Error(`refusing to use existing benchmark output directory: ${directory}`);
|
|
3095
|
+
throw error;
|
|
3096
|
+
}
|
|
3097
|
+
await syncDirectory(dirname(directory));
|
|
3098
|
+
}
|
|
3099
|
+
return {
|
|
3100
|
+
directory,
|
|
3101
|
+
initializationComplete: resolve(directory, ANALYST_BENCHMARK_INITIALIZATION_COMPLETE_FILE),
|
|
3102
|
+
manifest: resolve(directory, ANALYST_BENCHMARK_MANIFEST_FILE),
|
|
3103
|
+
observations: resolve(directory, ANALYST_BENCHMARK_OBSERVATIONS_FILE),
|
|
3104
|
+
costLedger: resolve(directory, ANALYST_BENCHMARK_COST_LEDGER_FILE),
|
|
3105
|
+
modelResponses: resolve(directory, "model-responses"),
|
|
3106
|
+
localReceipt: resolve(directory, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE),
|
|
3107
|
+
result: resolve(directory, "result.json"),
|
|
3108
|
+
report: resolve(directory, "report.md")
|
|
3109
|
+
};
|
|
3110
|
+
}
|
|
3111
|
+
async function prepareOutputLockPath(outDir) {
|
|
3112
|
+
const directory = resolve(outDir);
|
|
3113
|
+
await mkdir(dirname(directory), { recursive: true });
|
|
3114
|
+
return `${directory}.lock`;
|
|
3115
|
+
}
|
|
3116
|
+
function createRunIdentity(config, prepared) {
|
|
3117
|
+
const caseDefinitions = prepared.cases.map((testCase) => ({
|
|
3118
|
+
id: testCase.id,
|
|
3119
|
+
clusterId: testCase.clusterId,
|
|
3120
|
+
labelState: testCase.labelState,
|
|
3121
|
+
expectedIssues: testCase.expectedIssues,
|
|
3122
|
+
labeledEvidence: testCase.labeledEvidence ?? [],
|
|
3123
|
+
tags: testCase.tags ?? [],
|
|
3124
|
+
metadata: testCase.metadata ?? {}
|
|
3125
|
+
}));
|
|
3126
|
+
return {
|
|
3127
|
+
config: {
|
|
3128
|
+
dataset: config.dataset,
|
|
3129
|
+
datasetRevision: config.revision,
|
|
3130
|
+
datasetSplit: config.split,
|
|
3131
|
+
model: {
|
|
3132
|
+
id: config.model.model,
|
|
3133
|
+
maxOutputTokens: config.model.maxOutputTokens,
|
|
3134
|
+
timeoutMs: config.model.timeoutMs
|
|
3135
|
+
},
|
|
3136
|
+
limit: config.limit,
|
|
3137
|
+
seed: config.seed,
|
|
3138
|
+
concurrency: config.concurrency,
|
|
3139
|
+
repetitions: config.repetitions,
|
|
3140
|
+
maxCostUsd: config.maxCostUsd,
|
|
3141
|
+
maxArtifactBytes: config.maxArtifactBytes,
|
|
3142
|
+
analystProtocolSha256: publicBenchmarkProtocolSha256(config.dataset),
|
|
3143
|
+
implementationSha256: ANALYST_BENCHMARK_IMPLEMENTATION_SHA256,
|
|
3144
|
+
dependencyLockSha256: ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256,
|
|
3145
|
+
runnerIds: ["empty", "model"]
|
|
3146
|
+
},
|
|
3147
|
+
inputs: {
|
|
3148
|
+
labelsSha256: prepared.labelsSha256,
|
|
3149
|
+
sourceRowCount: prepared.sourceRowCount,
|
|
3150
|
+
selectedCaseIds: [...prepared.selectedCaseIds],
|
|
3151
|
+
traceFiles: prepared.traceFiles.map((traceFile) => ({ ...traceFile })),
|
|
3152
|
+
verificationArtifactsSha256: digestCanonical(prepared.verificationArtifacts),
|
|
3153
|
+
caseDefinitionsSha256: digestCanonical(caseDefinitions)
|
|
3154
|
+
}
|
|
3155
|
+
};
|
|
3156
|
+
}
|
|
3157
|
+
function createLocalRunReceipt(config, paths) {
|
|
3158
|
+
return {
|
|
3159
|
+
kind: "agent-eval/analyst-benchmark-local-run",
|
|
3160
|
+
local: {
|
|
3161
|
+
labelsPath: resolve(config.labelsPath),
|
|
3162
|
+
traceDir: resolve(config.traceDir),
|
|
3163
|
+
...config.artifactDir ? { artifactDir: resolve(config.artifactDir) } : {},
|
|
3164
|
+
outputDir: paths.directory,
|
|
3165
|
+
baseUrl: config.model.baseUrl,
|
|
3166
|
+
apiKeyEnvironment: config.apiKeyEnv
|
|
3167
|
+
},
|
|
3168
|
+
command: config.command,
|
|
3169
|
+
environment: {
|
|
3170
|
+
node: process.version,
|
|
3171
|
+
platform: platform(),
|
|
3172
|
+
arch: arch()
|
|
3173
|
+
},
|
|
3174
|
+
files: {
|
|
3175
|
+
manifest: paths.manifest,
|
|
3176
|
+
observations: paths.observations,
|
|
3177
|
+
costLedger: paths.costLedger,
|
|
3178
|
+
modelResponses: paths.modelResponses,
|
|
3179
|
+
result: paths.result,
|
|
3180
|
+
report: paths.report
|
|
3181
|
+
}
|
|
3182
|
+
};
|
|
3183
|
+
}
|
|
3184
|
+
async function initializeRunFiles(paths, identity, identitySha256, localIdentitySha256, localReceiptInput) {
|
|
3185
|
+
const existingManifest = await readOptionalRegularFile(paths.manifest, "benchmark run manifest");
|
|
3186
|
+
const manifest = existingManifest ? await readAndValidateManifestContent(paths.manifest, existingManifest, identity, identitySha256, localIdentitySha256) : {
|
|
3187
|
+
kind: "agent-eval/analyst-benchmark-run",
|
|
3188
|
+
createdAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
3189
|
+
identitySha256,
|
|
3190
|
+
localIdentitySha256,
|
|
3191
|
+
identity
|
|
3192
|
+
};
|
|
3193
|
+
const localReceipt = {
|
|
3194
|
+
...localReceiptInput,
|
|
3195
|
+
runIdentitySha256: identitySha256,
|
|
3196
|
+
localIdentitySha256
|
|
3197
|
+
};
|
|
3198
|
+
const manifestContent = `${JSON.stringify(manifest, null, 2)}\n`;
|
|
3199
|
+
const localReceiptContent = `${JSON.stringify(localReceipt, null, 2)}\n`;
|
|
3200
|
+
const initializationCompleteContent = renderInitializationComplete(manifest);
|
|
3201
|
+
await assertAbsentOrExact(paths.observations, "", "benchmark observation log");
|
|
3202
|
+
await assertAbsentOrExact(paths.localReceipt, localReceiptContent, "benchmark local run receipt");
|
|
3203
|
+
await assertAbsentOrExact(paths.manifest, manifestContent, "benchmark run manifest");
|
|
3204
|
+
for (const path of [
|
|
3205
|
+
paths.costLedger,
|
|
3206
|
+
paths.modelResponses,
|
|
3207
|
+
paths.result,
|
|
3208
|
+
paths.report
|
|
3209
|
+
]) if (await regularFileExists(path)) throw new Error(`benchmark initialization marker is missing but later run artifact exists: ${path}`);
|
|
3210
|
+
if (await regularFileExists(paths.initializationComplete)) throw new Error(`benchmark initialization marker already exists during partial initialization: ${paths.initializationComplete}`);
|
|
3211
|
+
await writeExclusiveOrVerify(paths.observations, "");
|
|
3212
|
+
await writeExclusiveOrVerify(paths.localReceipt, localReceiptContent);
|
|
3213
|
+
await writeExclusiveOrVerify(paths.manifest, manifestContent);
|
|
3214
|
+
await writeExclusiveOrVerify(paths.initializationComplete, initializationCompleteContent);
|
|
3215
|
+
return manifest;
|
|
3216
|
+
}
|
|
3217
|
+
async function readAndValidateResumeFiles(paths, currentIdentity, currentIdentitySha256, currentLocalIdentitySha256, localReceiptInput) {
|
|
3218
|
+
if (!await regularFileExists(paths.initializationComplete)) return initializeRunFiles(paths, currentIdentity, currentIdentitySha256, currentLocalIdentitySha256, localReceiptInput);
|
|
3219
|
+
const manifest = await readAndValidateManifest(paths.manifest, currentIdentity, currentIdentitySha256, currentLocalIdentitySha256);
|
|
3220
|
+
const localReceiptContent = await readRegularFile(paths.localReceipt, "benchmark local run receipt");
|
|
3221
|
+
const value = parseJson(localReceiptContent, paths.localReceipt);
|
|
3222
|
+
if (!isRecord$1(value)) throw new TypeError(`benchmark local run receipt must be an object: ${paths.localReceipt}`);
|
|
3223
|
+
assertExactKeys(value, [
|
|
3224
|
+
"kind",
|
|
3225
|
+
"runIdentitySha256",
|
|
3226
|
+
"localIdentitySha256",
|
|
3227
|
+
"local",
|
|
3228
|
+
"command",
|
|
3229
|
+
"environment",
|
|
3230
|
+
"files"
|
|
3231
|
+
], "benchmark local run receipt");
|
|
3232
|
+
if (value.kind !== "agent-eval/analyst-benchmark-local-run" || value.runIdentitySha256 !== currentIdentitySha256 || value.localIdentitySha256 !== currentLocalIdentitySha256 || !isRecord$1(value.local)) throw new Error("benchmark local run receipt does not match the requested resume");
|
|
3233
|
+
const expectedLocalReceipt = {
|
|
3234
|
+
...localReceiptInput,
|
|
3235
|
+
runIdentitySha256: currentIdentitySha256,
|
|
3236
|
+
localIdentitySha256: currentLocalIdentitySha256
|
|
3237
|
+
};
|
|
3238
|
+
if (digestCanonical(value.local) !== currentLocalIdentitySha256 || canonicalJson(value.local) !== canonicalJson(localReceiptInput.local)) throw new Error("benchmark local paths or endpoint do not match the requested resume");
|
|
3239
|
+
if (localReceiptContent !== `${JSON.stringify(expectedLocalReceipt, null, 2)}\n`) throw new Error(`benchmark local run receipt does not exactly match: ${paths.localReceipt}`);
|
|
3240
|
+
if (await readRegularFile(paths.initializationComplete, "benchmark initialization marker") !== renderInitializationComplete(manifest)) throw new Error(`benchmark initialization marker does not match the run manifest: ${paths.initializationComplete}`);
|
|
3241
|
+
return manifest;
|
|
3242
|
+
}
|
|
3243
|
+
async function readAndValidateManifest(path, currentIdentity, currentIdentitySha256, currentLocalIdentitySha256) {
|
|
3244
|
+
const content = await readRegularFile(path, "benchmark run manifest");
|
|
3245
|
+
const manifest = await readAndValidateManifestContent(path, content, currentIdentity, currentIdentitySha256, currentLocalIdentitySha256);
|
|
3246
|
+
if (content !== `${JSON.stringify(manifest, null, 2)}\n`) throw new Error(`benchmark run manifest does not exactly match: ${path}`);
|
|
3247
|
+
return manifest;
|
|
3248
|
+
}
|
|
3249
|
+
async function readAndValidateManifestContent(path, content, currentIdentity, currentIdentitySha256, currentLocalIdentitySha256) {
|
|
3250
|
+
const value = parseJson(content, path);
|
|
3251
|
+
if (!isRecord$1(value)) throw new TypeError(`benchmark run manifest must be an object: ${path}`);
|
|
3252
|
+
assertExactKeys(value, [
|
|
3253
|
+
"kind",
|
|
3254
|
+
"createdAt",
|
|
3255
|
+
"identitySha256",
|
|
3256
|
+
"localIdentitySha256",
|
|
3257
|
+
"identity"
|
|
3258
|
+
], "benchmark run manifest");
|
|
3259
|
+
if (value.kind !== "agent-eval/analyst-benchmark-run") throw new TypeError(`unsupported benchmark run manifest: ${path}`);
|
|
3260
|
+
if (typeof value.createdAt !== "string" || !Number.isFinite(Date.parse(value.createdAt))) throw new TypeError(`benchmark run manifest has an invalid createdAt: ${path}`);
|
|
3261
|
+
if (!isSha256(value.identitySha256) || !isSha256(value.localIdentitySha256) || !isRecord$1(value.identity)) throw new TypeError(`benchmark run manifest has an invalid identity: ${path}`);
|
|
3262
|
+
if (digestCanonical(value.identity) !== value.identitySha256) throw new Error(`benchmark run manifest identity digest does not match its contents: ${path}`);
|
|
3263
|
+
if (currentIdentitySha256 !== value.identitySha256 || currentLocalIdentitySha256 !== value.localIdentitySha256 || canonicalJson(currentIdentity) !== canonicalJson(value.identity)) throw new Error(`benchmark resume configuration or inputs do not match ${ANALYST_BENCHMARK_MANIFEST_FILE}`);
|
|
3264
|
+
return {
|
|
3265
|
+
kind: "agent-eval/analyst-benchmark-run",
|
|
3266
|
+
createdAt: value.createdAt,
|
|
3267
|
+
identitySha256: currentIdentitySha256,
|
|
3268
|
+
localIdentitySha256: currentLocalIdentitySha256,
|
|
3269
|
+
identity: currentIdentity
|
|
3270
|
+
};
|
|
3271
|
+
}
|
|
3272
|
+
function renderInitializationComplete(manifest) {
|
|
3273
|
+
return `${JSON.stringify({
|
|
3274
|
+
kind: "agent-eval/analyst-benchmark-initialization-complete",
|
|
3275
|
+
runIdentitySha256: manifest.identitySha256,
|
|
3276
|
+
localIdentitySha256: manifest.localIdentitySha256,
|
|
3277
|
+
createdAt: manifest.createdAt
|
|
3278
|
+
}, null, 2)}\n`;
|
|
3279
|
+
}
|
|
3280
|
+
async function assertAbsentOrExact(path, expected, label) {
|
|
3281
|
+
const existing = await readOptionalRegularFile(path, label);
|
|
3282
|
+
if (existing !== void 0 && existing !== expected) throw new Error(`${label} does not exactly match interrupted initialization: ${path}`);
|
|
3283
|
+
}
|
|
3284
|
+
async function readOptionalRegularFile(path, label) {
|
|
3285
|
+
if (!await regularFileExists(path)) return void 0;
|
|
3286
|
+
return readRegularFile(path, label);
|
|
3287
|
+
}
|
|
3288
|
+
function createObservationAppender(path, runIdentitySha256, progress) {
|
|
3289
|
+
let writes = Promise.resolve();
|
|
3290
|
+
const seen = new Set(progress.observations.map(observationKey));
|
|
3291
|
+
return (observation) => {
|
|
3292
|
+
const write = writes.then(async () => {
|
|
3293
|
+
assertAnalystBenchmarkObservation(observation, "benchmark observation");
|
|
3294
|
+
const key = observationKey(observation);
|
|
3295
|
+
if (seen.has(key)) throw new Error(`refusing duplicate benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}'`);
|
|
3296
|
+
const rowWithoutDigest = {
|
|
3297
|
+
sequence: progress.nextSequence,
|
|
3298
|
+
runIdentitySha256,
|
|
3299
|
+
previousRowSha256: progress.previousRowSha256,
|
|
3300
|
+
observation
|
|
3301
|
+
};
|
|
3302
|
+
const row = {
|
|
3303
|
+
...rowWithoutDigest,
|
|
3304
|
+
rowSha256: digestCanonical(rowWithoutDigest)
|
|
3305
|
+
};
|
|
3306
|
+
await appendDurable(path, `${JSON.stringify(row)}\n`);
|
|
3307
|
+
progress.nextSequence += 1;
|
|
3308
|
+
progress.previousRowSha256 = row.rowSha256;
|
|
3309
|
+
progress.observations.push(observation);
|
|
3310
|
+
seen.add(key);
|
|
3311
|
+
});
|
|
3312
|
+
writes = write;
|
|
3313
|
+
return write;
|
|
3314
|
+
};
|
|
3315
|
+
}
|
|
3316
|
+
async function readProgress(path, runIdentitySha256, caseIds, repetitions) {
|
|
3317
|
+
const rawLines = (await readRegularFile(path, "benchmark observation log")).split("\n");
|
|
3318
|
+
if (rawLines.at(-1) === "") rawLines.pop();
|
|
3319
|
+
const observations = [];
|
|
3320
|
+
const seen = /* @__PURE__ */ new Set();
|
|
3321
|
+
const executionIndexes = /* @__PURE__ */ new Set();
|
|
3322
|
+
let previousRowSha256 = null;
|
|
3323
|
+
const allowedCases = new Set(caseIds);
|
|
3324
|
+
const plannedObservationCount = caseIds.length * 2 * repetitions;
|
|
3325
|
+
for (const [index, line] of rawLines.entries()) {
|
|
3326
|
+
if (!line.trim()) throw new Error(`benchmark observation log contains an empty row at line ${index + 1}`);
|
|
3327
|
+
const parsed = parseJson(line, `${path}:${index + 1}`);
|
|
3328
|
+
if (!isRecord$1(parsed)) throw new TypeError(`benchmark observation row ${index + 1} must be an object`);
|
|
3329
|
+
assertExactKeys(parsed, [
|
|
3330
|
+
"sequence",
|
|
3331
|
+
"runIdentitySha256",
|
|
3332
|
+
"previousRowSha256",
|
|
3333
|
+
"observation",
|
|
3334
|
+
"rowSha256"
|
|
3335
|
+
], `benchmark observation row ${index + 1}`);
|
|
3336
|
+
assertAnalystBenchmarkObservation(parsed.observation, `benchmark observation row ${index + 1}.observation`);
|
|
3337
|
+
const observation = parsed.observation;
|
|
3338
|
+
const key = observationKey(observation);
|
|
3339
|
+
if (seen.has(key)) throw new Error(`duplicate benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' at line ${index + 1}`);
|
|
3340
|
+
if (parsed.sequence !== index) throw new Error(`benchmark observation row ${index + 1} has sequence ${String(parsed.sequence)}; expected ${index}`);
|
|
3341
|
+
if (parsed.runIdentitySha256 !== runIdentitySha256) throw new Error(`benchmark observation row ${index + 1} belongs to another run`);
|
|
3342
|
+
if (parsed.previousRowSha256 !== previousRowSha256) throw new Error(`benchmark observation row ${index + 1} breaks the digest chain`);
|
|
3343
|
+
if (!isSha256(parsed.rowSha256)) throw new TypeError(`benchmark observation row ${index + 1} has an invalid digest`);
|
|
3344
|
+
if (digestCanonical({
|
|
3345
|
+
sequence: parsed.sequence,
|
|
3346
|
+
runIdentitySha256: parsed.runIdentitySha256,
|
|
3347
|
+
previousRowSha256: parsed.previousRowSha256,
|
|
3348
|
+
observation
|
|
3349
|
+
}) !== parsed.rowSha256) throw new Error(`benchmark observation row ${index + 1} digest does not match its contents`);
|
|
3350
|
+
if (!allowedCases.has(observation.caseId) || observation.runnerId !== "empty" && observation.runnerId !== "model" || observation.repetition >= repetitions || observation.executionIndex >= plannedObservationCount) throw new Error(`benchmark observation row ${index + 1} does not match a planned case, runner, and repetition`);
|
|
3351
|
+
if (executionIndexes.has(observation.executionIndex)) throw new Error(`duplicate benchmark executionIndex ${observation.executionIndex} at line ${index + 1}`);
|
|
3352
|
+
observations.push(observation);
|
|
3353
|
+
seen.add(key);
|
|
3354
|
+
executionIndexes.add(observation.executionIndex);
|
|
3355
|
+
previousRowSha256 = parsed.rowSha256;
|
|
3356
|
+
}
|
|
3357
|
+
return {
|
|
3358
|
+
observations,
|
|
3359
|
+
nextSequence: observations.length,
|
|
3360
|
+
previousRowSha256
|
|
3361
|
+
};
|
|
3362
|
+
}
|
|
3363
|
+
async function writeExclusiveOrVerify(path, content) {
|
|
3364
|
+
try {
|
|
3365
|
+
await writeExclusive(path, content);
|
|
3366
|
+
} catch (error) {
|
|
3367
|
+
if (!isNodeError(error, "EEXIST")) throw error;
|
|
3368
|
+
if (await readRegularFile(path, "existing benchmark artifact") !== content) throw new Error(`refusing to replace existing benchmark artifact: ${path}`);
|
|
3369
|
+
}
|
|
3370
|
+
}
|
|
3371
|
+
async function regularFileExists(path) {
|
|
3372
|
+
try {
|
|
3373
|
+
const fileStat = await lstat(path);
|
|
3374
|
+
if (!fileStat.isFile() || fileStat.isSymbolicLink()) throw new Error(`benchmark artifact path must be a real file: ${path}`);
|
|
3375
|
+
return true;
|
|
3376
|
+
} catch (error) {
|
|
3377
|
+
if (isNodeError(error, "ENOENT")) return false;
|
|
3378
|
+
throw error;
|
|
3379
|
+
}
|
|
3380
|
+
}
|
|
3381
|
+
async function appendDurable(path, content) {
|
|
3382
|
+
const handle = await open(path, constants.O_APPEND | constants.O_WRONLY | constants.O_NOFOLLOW);
|
|
3383
|
+
try {
|
|
3384
|
+
await handle.writeFile(content, "utf8");
|
|
3385
|
+
await handle.sync();
|
|
3386
|
+
} finally {
|
|
3387
|
+
await handle.close();
|
|
3388
|
+
}
|
|
3389
|
+
}
|
|
3390
|
+
async function writeExclusive(path, content) {
|
|
3391
|
+
const temporary = `${path}.tmp-${process.pid}-${randomUUID()}`;
|
|
3392
|
+
let handle;
|
|
3393
|
+
try {
|
|
3394
|
+
handle = await open(temporary, "wx");
|
|
3395
|
+
await handle.writeFile(content, "utf8");
|
|
3396
|
+
await handle.sync();
|
|
3397
|
+
await handle.close();
|
|
3398
|
+
handle = void 0;
|
|
3399
|
+
await link(temporary, path);
|
|
3400
|
+
await syncDirectory(dirname(path));
|
|
3401
|
+
} finally {
|
|
3402
|
+
await handle?.close().catch(() => void 0);
|
|
3403
|
+
await unlink(temporary).catch(() => void 0);
|
|
3404
|
+
}
|
|
3405
|
+
}
|
|
3406
|
+
async function readRegularFile(path, label) {
|
|
3407
|
+
let fileStat;
|
|
3408
|
+
try {
|
|
3409
|
+
fileStat = await lstat(path);
|
|
3410
|
+
} catch (error) {
|
|
3411
|
+
if (isNodeError(error, "ENOENT")) throw new Error(`${label} is missing: ${path}`);
|
|
3412
|
+
throw error;
|
|
3413
|
+
}
|
|
3414
|
+
if (!fileStat.isFile() || fileStat.isSymbolicLink()) throw new Error(`${label} must be a real file: ${path}`);
|
|
3415
|
+
return readFile(path, "utf8");
|
|
3416
|
+
}
|
|
3417
|
+
async function syncDirectory(path) {
|
|
3418
|
+
const directory = await open(path, "r");
|
|
3419
|
+
try {
|
|
3420
|
+
await directory.sync();
|
|
3421
|
+
} finally {
|
|
3422
|
+
await directory.close();
|
|
3423
|
+
}
|
|
3424
|
+
}
|
|
3425
|
+
function isNodeError(error, code) {
|
|
3426
|
+
return error instanceof Error && "code" in error && error.code === code;
|
|
3427
|
+
}
|
|
3428
|
+
//#endregion
|
|
3429
|
+
//#region src/analyst/benchmark-comparison.ts
|
|
3430
|
+
function compareAnalystRunners(result, options) {
|
|
3431
|
+
const confidence = options.confidence ?? .95;
|
|
3432
|
+
const resamples = options.resamples ?? 2e3;
|
|
3433
|
+
assertComparisonControls(confidence, resamples);
|
|
3434
|
+
const runnerIds = new Set(result.summaries.map((summary) => summary.runnerId));
|
|
3435
|
+
if (!runnerIds.has(options.baselineRunnerId)) throw new TypeError(`unknown baseline analyst runner '${options.baselineRunnerId}'`);
|
|
3436
|
+
if (!runnerIds.has(options.candidateRunnerId)) throw new TypeError(`unknown candidate analyst runner '${options.candidateRunnerId}'`);
|
|
3437
|
+
if (options.baselineRunnerId === options.candidateRunnerId) throw new TypeError("baseline and candidate analyst runners must be different");
|
|
3438
|
+
const baseline = observationsByCase(result.observations, options.baselineRunnerId);
|
|
3439
|
+
const candidate = observationsByCase(result.observations, options.candidateRunnerId);
|
|
3440
|
+
const populationRepresentativenessProven = result.provenance.metadata?.populationRepresentativenessProven === true;
|
|
3441
|
+
const metrics = METRICS.map((metric) => compareMetric({
|
|
3442
|
+
metric,
|
|
3443
|
+
baseline,
|
|
3444
|
+
candidate,
|
|
3445
|
+
confidence,
|
|
3446
|
+
resamples,
|
|
3447
|
+
seed: options.seed,
|
|
3448
|
+
populationRepresentativenessProven
|
|
3449
|
+
}));
|
|
3450
|
+
return {
|
|
3451
|
+
baselineRunnerId: options.baselineRunnerId,
|
|
3452
|
+
candidateRunnerId: options.candidateRunnerId,
|
|
3453
|
+
metrics
|
|
3454
|
+
};
|
|
3455
|
+
}
|
|
3456
|
+
function compareMetric(options) {
|
|
3457
|
+
const pairedCases = [];
|
|
3458
|
+
let eligibleObservations = 0;
|
|
3459
|
+
let pairedObservations = 0;
|
|
3460
|
+
let baselineMissingObservations = 0;
|
|
3461
|
+
let candidateMissingObservations = 0;
|
|
3462
|
+
let asymmetricMissingObservations = 0;
|
|
3463
|
+
const caseIds = /* @__PURE__ */ new Set([...options.baseline.keys(), ...options.candidate.keys()]);
|
|
3464
|
+
for (const caseId of caseIds) {
|
|
3465
|
+
const baselineByRepetition = new Map((options.baseline.get(caseId) ?? []).map((observation) => [observation.repetition, observation]));
|
|
3466
|
+
const candidateByRepetition = new Map((options.candidate.get(caseId) ?? []).map((observation) => [observation.repetition, observation]));
|
|
3467
|
+
const caseBefore = [];
|
|
3468
|
+
const caseAfter = [];
|
|
3469
|
+
let clusterId;
|
|
3470
|
+
const repetitions = /* @__PURE__ */ new Set([...baselineByRepetition.keys(), ...candidateByRepetition.keys()]);
|
|
3471
|
+
for (const repetition of repetitions) {
|
|
3472
|
+
const baselineObservation = baselineByRepetition.get(repetition);
|
|
3473
|
+
const candidateObservation = candidateByRepetition.get(repetition);
|
|
3474
|
+
const identity = baselineObservation ?? candidateObservation;
|
|
3475
|
+
if (!identity || !metricApplies(identity, options.metric)) continue;
|
|
3476
|
+
if (baselineObservation && candidateObservation) assertSameCaseIdentity(baselineObservation, candidateObservation);
|
|
3477
|
+
eligibleObservations += 1;
|
|
3478
|
+
clusterId = identity.clusterId;
|
|
3479
|
+
const baselineValue = baselineObservation ? metricValue(baselineObservation, options.metric) : null;
|
|
3480
|
+
const candidateValue = candidateObservation ? metricValue(candidateObservation, options.metric) : null;
|
|
3481
|
+
const baselineMissing = baselineValue === null;
|
|
3482
|
+
const candidateMissing = candidateValue === null;
|
|
3483
|
+
if (baselineMissing) baselineMissingObservations += 1;
|
|
3484
|
+
if (candidateMissing) candidateMissingObservations += 1;
|
|
3485
|
+
if (baselineMissing !== candidateMissing) asymmetricMissingObservations += 1;
|
|
3486
|
+
if (baselineMissing || candidateMissing) continue;
|
|
3487
|
+
caseBefore.push(baselineValue);
|
|
3488
|
+
caseAfter.push(candidateValue);
|
|
3489
|
+
pairedObservations += 1;
|
|
3490
|
+
}
|
|
3491
|
+
if (caseBefore.length === 0 || !clusterId) continue;
|
|
3492
|
+
pairedCases.push({
|
|
3493
|
+
clusterId,
|
|
3494
|
+
baseline: mean$1(caseBefore),
|
|
3495
|
+
candidate: mean$1(caseAfter)
|
|
3496
|
+
});
|
|
3497
|
+
}
|
|
3498
|
+
const byCluster = /* @__PURE__ */ new Map();
|
|
3499
|
+
for (const pairedCase of pairedCases) {
|
|
3500
|
+
const rows = byCluster.get(pairedCase.clusterId) ?? [];
|
|
3501
|
+
rows.push(pairedCase);
|
|
3502
|
+
byCluster.set(pairedCase.clusterId, rows);
|
|
3503
|
+
}
|
|
3504
|
+
const before = [...byCluster.values()].map((rows) => mean$1(rows.map((row) => row.baseline)));
|
|
3505
|
+
const after = [...byCluster.values()].map((rows) => mean$1(rows.map((row) => row.candidate)));
|
|
3506
|
+
const interval = before.length === 0 ? null : pairedBootstrap(before, after, {
|
|
3507
|
+
confidence: options.confidence,
|
|
3508
|
+
resamples: options.resamples,
|
|
3509
|
+
statistic: "mean",
|
|
3510
|
+
seed: options.seed
|
|
3511
|
+
});
|
|
3512
|
+
const survivorOnly = pairedObservations < eligibleObservations;
|
|
3513
|
+
const limitations = [];
|
|
3514
|
+
if (!interval?.gateEligible) limitations.push("fewer-than-20-independent-clusters");
|
|
3515
|
+
if (!options.populationRepresentativenessProven) limitations.push("population-representativeness-not-proven");
|
|
3516
|
+
if (survivorOnly) limitations.push("missing-observations");
|
|
3517
|
+
const comparison = {
|
|
3518
|
+
metric: options.metric,
|
|
3519
|
+
direction: LOWER_IS_BETTER.has(options.metric) ? "lower" : "higher",
|
|
3520
|
+
pairedCases: pairedCases.length,
|
|
3521
|
+
pairedClusters: before.length,
|
|
3522
|
+
eligibleObservations,
|
|
3523
|
+
pairedObservations,
|
|
3524
|
+
baselineMissingObservations,
|
|
3525
|
+
candidateMissingObservations,
|
|
3526
|
+
asymmetricMissingObservations,
|
|
3527
|
+
survivorOnly,
|
|
3528
|
+
baselineMean: before.length === 0 ? null : mean$1(before),
|
|
3529
|
+
candidateMean: after.length === 0 ? null : mean$1(after),
|
|
3530
|
+
meanDelta: interval?.mean ?? null,
|
|
3531
|
+
intervalLow: interval?.low ?? null,
|
|
3532
|
+
intervalHigh: interval?.high ?? null,
|
|
3533
|
+
confidence: options.confidence,
|
|
3534
|
+
resamples: options.resamples,
|
|
3535
|
+
minimumSampleMet: interval?.gateEligible ?? false,
|
|
3536
|
+
populationInferenceEligible: limitations.length === 0,
|
|
3537
|
+
inferenceLimitations: limitations
|
|
3538
|
+
};
|
|
3539
|
+
assertValidComparison(comparison);
|
|
3540
|
+
return comparison;
|
|
3541
|
+
}
|
|
3542
|
+
const METRICS = [
|
|
3543
|
+
"completion",
|
|
3544
|
+
"issueRecall",
|
|
3545
|
+
"findingPrecision",
|
|
3546
|
+
"f1",
|
|
3547
|
+
"criticalStepAccuracy",
|
|
3548
|
+
"citationCoverage",
|
|
3549
|
+
"citationExcerptCoverage",
|
|
3550
|
+
"citationLabelAgreement",
|
|
3551
|
+
"citationResolution",
|
|
3552
|
+
"trustedNegativeAccuracy",
|
|
3553
|
+
"latencyMs",
|
|
3554
|
+
"calls",
|
|
3555
|
+
"inputTokens",
|
|
3556
|
+
"outputTokens",
|
|
3557
|
+
"reasoningTokens",
|
|
3558
|
+
"cachedTokens",
|
|
3559
|
+
"cacheWriteTokens",
|
|
3560
|
+
"costUsd"
|
|
3561
|
+
];
|
|
3562
|
+
const LOWER_IS_BETTER = /* @__PURE__ */ new Set([
|
|
3563
|
+
"latencyMs",
|
|
3564
|
+
"calls",
|
|
3565
|
+
"inputTokens",
|
|
3566
|
+
"outputTokens",
|
|
3567
|
+
"reasoningTokens",
|
|
3568
|
+
"cachedTokens",
|
|
3569
|
+
"cacheWriteTokens",
|
|
3570
|
+
"costUsd"
|
|
3571
|
+
]);
|
|
3572
|
+
function observationsByCase(observations, runnerId) {
|
|
3573
|
+
const byCase = /* @__PURE__ */ new Map();
|
|
3574
|
+
for (const observation of observations) {
|
|
3575
|
+
if (observation.runnerId !== runnerId) continue;
|
|
3576
|
+
const rows = byCase.get(observation.caseId) ?? [];
|
|
3577
|
+
rows.push(observation);
|
|
3578
|
+
byCase.set(observation.caseId, rows);
|
|
3579
|
+
}
|
|
3580
|
+
return byCase;
|
|
3581
|
+
}
|
|
3582
|
+
function assertSameCaseIdentity(baseline, candidate) {
|
|
3583
|
+
if (baseline.clusterId !== candidate.clusterId || baseline.labelState !== candidate.labelState) throw new Error(`analyst comparison case identity differs for '${baseline.caseId}' repetition ${baseline.repetition}`);
|
|
3584
|
+
}
|
|
3585
|
+
function metricApplies(observation, metric) {
|
|
3586
|
+
if (metric === "trustedNegativeAccuracy") return observation.labelState === "trusted-negative";
|
|
3587
|
+
if (metric === "issueRecall" || metric === "findingPrecision" || metric === "f1") return observation.labelState === "positive";
|
|
3588
|
+
if (metric === "criticalStepAccuracy") return observation.labelState === "positive" && observation.score.criticalStepAccuracy !== null;
|
|
3589
|
+
return true;
|
|
3590
|
+
}
|
|
3591
|
+
function metricValue(observation, metric) {
|
|
3592
|
+
if (metric === "completion") return observation.error ? 0 : 1;
|
|
3593
|
+
if (metric === "latencyMs") return observation.latencyMs;
|
|
3594
|
+
if (metric === "trustedNegativeAccuracy") {
|
|
3595
|
+
if (observation.error) return 0;
|
|
3596
|
+
return observation.score.predictionOnLabelEmptyCase ? 0 : 1;
|
|
3597
|
+
}
|
|
3598
|
+
if (observation.error && (metric === "issueRecall" || metric === "findingPrecision" || metric === "f1" || metric === "criticalStepAccuracy")) return 0;
|
|
3599
|
+
if (observation.error && (metric === "citationCoverage" || metric === "citationExcerptCoverage" || metric === "citationLabelAgreement" || metric === "citationResolution")) return null;
|
|
3600
|
+
if (metric === "issueRecall") return observation.score.issueRecall;
|
|
3601
|
+
if (metric === "findingPrecision") return observation.score.findingPrecision;
|
|
3602
|
+
if (metric === "f1") return observation.score.f1;
|
|
3603
|
+
if (metric === "criticalStepAccuracy") return observation.score.criticalStepAccuracy;
|
|
3604
|
+
if (metric === "citationCoverage") return observation.score.citationCoverage;
|
|
3605
|
+
if (metric === "citationExcerptCoverage") return observation.score.citationExcerptCoverage;
|
|
3606
|
+
if (metric === "citationLabelAgreement") return observation.score.citationLabelAgreement;
|
|
3607
|
+
if (metric === "citationResolution") return observation.evidenceResolution?.validity ?? null;
|
|
3608
|
+
if (metric === "calls") return observation.usage?.calls ?? null;
|
|
3609
|
+
if (metric === "inputTokens") return observation.usage?.tokens?.input ?? null;
|
|
3610
|
+
if (metric === "outputTokens") return observation.usage?.tokens?.output ?? null;
|
|
3611
|
+
if (metric === "reasoningTokens") return observation.usage?.tokens?.reasoning ?? null;
|
|
3612
|
+
if (metric === "cachedTokens") return observation.usage?.tokens?.cached ?? null;
|
|
3613
|
+
if (metric === "cacheWriteTokens") return observation.usage?.tokens?.cacheWrite ?? null;
|
|
3614
|
+
if (observation.usage?.cost.kind === "uncaptured") return null;
|
|
3615
|
+
return observation.usage?.cost.usd ?? null;
|
|
3616
|
+
}
|
|
3617
|
+
function mean$1(values) {
|
|
3618
|
+
return values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
3619
|
+
}
|
|
3620
|
+
function assertComparisonControls(confidence, resamples) {
|
|
3621
|
+
if (!Number.isSafeInteger(resamples) || resamples <= 0 || resamples > 1e6) throw new Error(`compareAnalystRunners: resamples must be a positive safe integer no greater than 1000000, got ${String(resamples)}`);
|
|
3622
|
+
if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new Error(`compareAnalystRunners: confidence must be a finite number in (0,1), got ${String(confidence)}`);
|
|
3623
|
+
}
|
|
3624
|
+
function assertValidComparison(comparison) {
|
|
3625
|
+
if ([
|
|
3626
|
+
"pairedCases",
|
|
3627
|
+
"pairedClusters",
|
|
3628
|
+
"eligibleObservations",
|
|
3629
|
+
"pairedObservations",
|
|
3630
|
+
"baselineMissingObservations",
|
|
3631
|
+
"candidateMissingObservations",
|
|
3632
|
+
"asymmetricMissingObservations",
|
|
3633
|
+
"confidence",
|
|
3634
|
+
"resamples"
|
|
3635
|
+
].some((field) => !Number.isFinite(comparison[field])) || [
|
|
3636
|
+
"baselineMean",
|
|
3637
|
+
"candidateMean",
|
|
3638
|
+
"meanDelta",
|
|
3639
|
+
"intervalLow",
|
|
3640
|
+
"intervalHigh"
|
|
3641
|
+
].some((field) => comparison[field] !== null && !Number.isFinite(comparison[field]))) throw new Error(`compareAnalystRunners: ${comparison.metric} produced non-finite comparison output`);
|
|
3642
|
+
if (comparison.intervalLow !== null && comparison.intervalHigh !== null && comparison.intervalLow > comparison.intervalHigh) throw new Error(`compareAnalystRunners: ${comparison.metric} produced an invalid confidence interval`);
|
|
3643
|
+
}
|
|
3644
|
+
//#endregion
|
|
3645
|
+
//#region src/analyst/benchmark-public-calibration.ts
|
|
3646
|
+
function summarizeCodeTraceCalibration(result) {
|
|
3647
|
+
return {
|
|
3648
|
+
protocol: "labeled-positive-and-solved-negative",
|
|
3649
|
+
rationale: "Uses rows with incorrect-step labels as positives and solved label-empty rows as trusted negatives. Failed label-empty rows remain in the published result but are not treated as clean controls.",
|
|
3650
|
+
runners: result.provenance.runnerIds.map((runnerId) => summarizeRunner(runnerId, result.observations.filter((observation) => observation.runnerId === runnerId)))
|
|
3651
|
+
};
|
|
3652
|
+
}
|
|
3653
|
+
function renderCodeTraceCalibrationMarkdown(summary) {
|
|
3654
|
+
return [
|
|
3655
|
+
"## CodeTraceBench Calibrated View",
|
|
3656
|
+
"",
|
|
3657
|
+
summary.rationale,
|
|
3658
|
+
"",
|
|
3659
|
+
"| Runner | Completed/selected | Failed | Positive runs | Trusted negative runs | Unlabeled runs | Failed label-empty | Unknown label-empty | Matched/expected steps | Predicted steps | Precision | Recall | F1 | Official all-row F1 | Official rows | Trusted-negative false positives | Trusted-negative failures | Unlabeled predictions | Unlabeled failures |",
|
|
3660
|
+
"| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |",
|
|
3661
|
+
...summary.runners.map((runner) => `| ${escapeCell$1(runner.runnerId)} | ${runner.completedRuns}/${runner.selectedRuns} | ${runner.failedRuns} | ${runner.positiveRuns} | ${runner.trustedNegativeRuns} | ${runner.unlabeledRuns} | ${runner.failedLabelEmptyRuns} | ${runner.unknownLabelEmptyRuns} | ${runner.matchedIncorrectSteps}/${runner.expectedIncorrectSteps} | ${runner.predictedIncorrectSteps} | ${rate$1(runner.precision)} | ${rate$1(runner.recall)} | ${rate$1(runner.f1)} | ${rate$1(runner.officialAllRowF1)} | ${runner.officialAllRowRuns} | ${rate$1(runner.trustedNegativeFalsePositiveRate)} | ${rate$1(runner.trustedNegativeFailureRate)} | ${rate$1(runner.unlabeledPredictionRate)} | ${rate$1(runner.unlabeledFailureRate)} |`)
|
|
3662
|
+
].join("\n");
|
|
3663
|
+
}
|
|
3664
|
+
function summarizeRunner(runnerId, observations) {
|
|
3665
|
+
const positive = observations.filter((observation) => observation.labelState === "positive");
|
|
3666
|
+
const trustedNegative = observations.filter((observation) => observation.labelState === "trusted-negative");
|
|
3667
|
+
const excluded = observations.filter((observation) => observation.labelState === "unlabeled");
|
|
3668
|
+
const selected = [...positive, ...trustedNegative];
|
|
3669
|
+
const expected = sum(positive.map((observation) => observation.score.expectedIssueCount));
|
|
3670
|
+
const predicted = sum(selected.map((observation) => observation.error ? 0 : observation.findings.length));
|
|
3671
|
+
const matched = sum(positive.map((observation) => observation.score.matchedIssueIds.length));
|
|
3672
|
+
const precision = predicted === 0 ? expected > 0 ? 0 : null : matched / predicted;
|
|
3673
|
+
const recall = ratio(matched, expected);
|
|
3674
|
+
const completedTrustedNegative = trustedNegative.filter((observation) => !observation.error);
|
|
3675
|
+
const completedExcluded = excluded.filter((observation) => !observation.error);
|
|
3676
|
+
const officialRows = observations.map(officialCodeTraceF1);
|
|
3677
|
+
return {
|
|
3678
|
+
runnerId,
|
|
3679
|
+
selectedRuns: selected.length,
|
|
3680
|
+
positiveRuns: positive.length,
|
|
3681
|
+
trustedNegativeRuns: trustedNegative.length,
|
|
3682
|
+
unlabeledRuns: excluded.length,
|
|
3683
|
+
failedLabelEmptyRuns: excluded.filter((observation) => observation.caseMetadata?.solved === false).length,
|
|
3684
|
+
unknownLabelEmptyRuns: excluded.filter((observation) => observation.caseMetadata?.solved !== false).length,
|
|
3685
|
+
completedRuns: selected.filter((observation) => !observation.error).length,
|
|
3686
|
+
failedRuns: selected.filter((observation) => observation.error).length,
|
|
3687
|
+
expectedIncorrectSteps: expected,
|
|
3688
|
+
predictedIncorrectSteps: predicted,
|
|
3689
|
+
matchedIncorrectSteps: matched,
|
|
3690
|
+
officialAllRowF1: mean(officialRows),
|
|
3691
|
+
officialAllRowRuns: officialRows.length,
|
|
3692
|
+
precision,
|
|
3693
|
+
recall,
|
|
3694
|
+
f1: harmonicMean(precision, recall),
|
|
3695
|
+
trustedNegativeFalsePositiveRate: ratio(completedTrustedNegative.filter((observation) => observation.score.predictionOnLabelEmptyCase).length, completedTrustedNegative.length),
|
|
3696
|
+
trustedNegativeFailureRate: ratio(trustedNegative.filter((observation) => observation.error).length, trustedNegative.length),
|
|
3697
|
+
unlabeledPredictionRate: ratio(completedExcluded.filter((observation) => observation.findings.length > 0).length, completedExcluded.length),
|
|
3698
|
+
unlabeledFailureRate: ratio(excluded.filter((observation) => Boolean(observation.error)).length, excluded.length)
|
|
3699
|
+
};
|
|
3700
|
+
}
|
|
3701
|
+
function officialCodeTraceF1(observation) {
|
|
3702
|
+
const trajectoryId = observation.caseMetadata?.trajectoryId;
|
|
3703
|
+
if (typeof trajectoryId !== "string" || !trajectoryId.trim()) throw new TypeError(`${observation.caseId}: CodeTraceBench trajectoryId metadata is missing`);
|
|
3704
|
+
const expected = new Set([...observation.score.matchedIssueIds, ...observation.score.missedIssueIds].map((issueId) => {
|
|
3705
|
+
const match = /^incorrect:(\d+)$/.exec(issueId);
|
|
3706
|
+
if (!match) throw new TypeError(`${observation.caseId}: invalid incorrect-step label '${issueId}'`);
|
|
3707
|
+
return Number(match[1]);
|
|
3708
|
+
}));
|
|
3709
|
+
const predicted = /* @__PURE__ */ new Set();
|
|
3710
|
+
if (!observation.error) for (const finding of observation.findings) {
|
|
3711
|
+
if (finding.area !== "incorrect") continue;
|
|
3712
|
+
for (const evidence of finding.evidence_refs) {
|
|
3713
|
+
const location = codeTraceStepFromEvidence(evidence.uri);
|
|
3714
|
+
if (!location || location.traceId !== trajectoryId) throw new TypeError(`${observation.caseId}: invalid CodeTraceBench prediction evidence '${evidence.uri}'`);
|
|
3715
|
+
predicted.add(location.step);
|
|
3716
|
+
}
|
|
3717
|
+
}
|
|
3718
|
+
let matched = 0;
|
|
3719
|
+
for (const step of predicted) if (expected.has(step)) matched += 1;
|
|
3720
|
+
return harmonicMean(predicted.size === 0 ? 0 : matched / predicted.size, expected.size === 0 ? 0 : matched / expected.size) ?? 0;
|
|
3721
|
+
}
|
|
3722
|
+
function sum(values) {
|
|
3723
|
+
return values.reduce((total, value) => total + value, 0);
|
|
3724
|
+
}
|
|
3725
|
+
function ratio(numerator, denominator) {
|
|
3726
|
+
return denominator === 0 ? null : numerator / denominator;
|
|
3727
|
+
}
|
|
3728
|
+
function mean(values) {
|
|
3729
|
+
return values.length === 0 ? null : sum(values) / values.length;
|
|
3730
|
+
}
|
|
3731
|
+
function harmonicMean(left, right) {
|
|
3732
|
+
if (left === null || right === null) return null;
|
|
3733
|
+
return left + right === 0 ? 0 : 2 * left * right / (left + right);
|
|
3734
|
+
}
|
|
3735
|
+
function rate$1(value) {
|
|
3736
|
+
return value === null ? "n/a" : value.toFixed(3);
|
|
3737
|
+
}
|
|
3738
|
+
function escapeCell$1(value) {
|
|
3739
|
+
return value.replaceAll("|", "\\|").replaceAll("\n", " ");
|
|
3740
|
+
}
|
|
3741
|
+
//#endregion
|
|
3742
|
+
//#region src/analyst/benchmark-command-result.ts
|
|
3743
|
+
async function readAnalystBenchmarkArtifact(path) {
|
|
3744
|
+
const value = parseJson(await readRegularFile(path, "analyst benchmark result"), path);
|
|
3745
|
+
assertAnalystBenchmarkArtifact(value, "analyst benchmark result");
|
|
3746
|
+
return value;
|
|
3747
|
+
}
|
|
3748
|
+
function assertCompletedArtifactMatchesRun(artifact, manifest, observations, prepared) {
|
|
3749
|
+
if (artifact.runIdentitySha256 !== manifest.identitySha256) throw new Error("completed benchmark result belongs to another run");
|
|
3750
|
+
assertSameObservations(artifact.result.observations, observations);
|
|
3751
|
+
const expectedCount = manifest.identity.inputs.selectedCaseIds.length * manifest.identity.config.runnerIds.length * manifest.identity.config.repetitions;
|
|
3752
|
+
if (observations.length !== expectedCount) throw new Error(`completed benchmark result has ${observations.length} observations; expected ${expectedCount}`);
|
|
3753
|
+
const { config, inputs } = manifest.identity;
|
|
3754
|
+
const verificationAvailability = {
|
|
3755
|
+
cases: prepared.verificationArtifacts.length,
|
|
3756
|
+
resultFilesPresent: prepared.verificationArtifacts.filter((artifact) => artifact.status === "present").length,
|
|
3757
|
+
resultFilesMissing: prepared.verificationArtifacts.filter((artifact) => artifact.status === "missing").length,
|
|
3758
|
+
outcomes: {
|
|
3759
|
+
passed: prepared.verificationArtifacts.filter((artifact) => artifact.outcome.status === "passed").length,
|
|
3760
|
+
failed: prepared.verificationArtifacts.filter((artifact) => artifact.outcome.status === "failed").length,
|
|
3761
|
+
unavailable: prepared.verificationArtifacts.filter((artifact) => artifact.outcome.status === "unavailable").length
|
|
3762
|
+
}
|
|
3763
|
+
};
|
|
3764
|
+
const expectedInputs = {
|
|
3765
|
+
dataset: config.dataset,
|
|
3766
|
+
datasetRevision: config.datasetRevision,
|
|
3767
|
+
datasetSplit: config.datasetSplit,
|
|
3768
|
+
labelsSha256: inputs.labelsSha256,
|
|
3769
|
+
sourceRowCount: inputs.sourceRowCount,
|
|
3770
|
+
traceFiles: inputs.traceFiles.map((traceFile) => ({ ...traceFile })),
|
|
3771
|
+
verificationArtifacts: prepared.verificationArtifacts,
|
|
3772
|
+
verificationAvailability,
|
|
3773
|
+
selection: {
|
|
3774
|
+
limit: config.limit,
|
|
3775
|
+
seed: config.seed,
|
|
3776
|
+
selectedCaseIds: [...inputs.selectedCaseIds],
|
|
3777
|
+
report: prepared.selection
|
|
3778
|
+
},
|
|
3779
|
+
execution: {
|
|
3780
|
+
repetitions: config.repetitions,
|
|
3781
|
+
concurrency: config.concurrency,
|
|
3782
|
+
model: config.model.id,
|
|
3783
|
+
maxOutputTokens: config.model.maxOutputTokens,
|
|
3784
|
+
timeoutMs: config.model.timeoutMs,
|
|
3785
|
+
maxCostUsd: config.maxCostUsd,
|
|
3786
|
+
maxArtifactBytes: config.maxArtifactBytes,
|
|
3787
|
+
analystProtocolSha256: config.analystProtocolSha256,
|
|
3788
|
+
implementationSha256: config.implementationSha256,
|
|
3789
|
+
dependencyLockSha256: config.dependencyLockSha256
|
|
3790
|
+
}
|
|
3791
|
+
};
|
|
3792
|
+
if (canonicalJson(artifact.inputs) !== canonicalJson(expectedInputs)) throw new Error("completed benchmark result inputs do not match the run manifest");
|
|
3793
|
+
const provenance = artifact.result.provenance;
|
|
3794
|
+
const expectedDatasetId = config.dataset === "agentrx" ? "microsoft/AgentRx" : "NJU-LINK/CodeTraceBench";
|
|
3795
|
+
const expectedOutputAdapter = config.dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-step";
|
|
3796
|
+
if (provenance.id !== `${config.dataset}-real-model-analyst` || provenance.startedAt !== manifest.createdAt || !Number.isFinite(Date.parse(provenance.endedAt)) || Date.parse(provenance.endedAt) < Date.parse(provenance.startedAt) || canonicalJson(provenance.dataset) !== canonicalJson({
|
|
3797
|
+
id: expectedDatasetId,
|
|
3798
|
+
revision: config.datasetRevision,
|
|
3799
|
+
split: config.datasetSplit
|
|
3800
|
+
}) || provenance.caseCount !== inputs.selectedCaseIds.length || canonicalJson(provenance.runnerIds) !== canonicalJson(config.runnerIds) || provenance.repetitions !== config.repetitions || provenance.maxConcurrency !== Math.min(config.concurrency, expectedCount) || provenance.runnerOrderSeed !== config.seed || provenance.metadata?.model !== config.model.id || provenance.metadata?.outputAdapter !== expectedOutputAdapter || provenance.metadata?.caseSelection !== prepared.selection.method || provenance.metadata?.caseSelectionSeed !== config.seed || provenance.metadata?.selectionStratified !== prepared.selection.stratified || provenance.metadata?.protocolSha256 !== config.analystProtocolSha256 || provenance.metadata?.implementationSha256 !== config.implementationSha256 || provenance.metadata?.dependencyLockSha256 !== config.dependencyLockSha256 || provenance.metadata?.populationRepresentativenessProven !== false) throw new Error("completed benchmark result provenance does not match the run manifest");
|
|
3801
|
+
const expectedSummaries = config.runnerIds.map((runnerId) => summarizeAnalystBenchmarkRunner(runnerId, observations.filter((observation) => observation.runnerId === runnerId)));
|
|
3802
|
+
if (canonicalJson(artifact.result.summaries) !== canonicalJson(expectedSummaries)) throw new Error("completed benchmark summaries do not match durable observations");
|
|
3803
|
+
const expectedComparisons = [compareAnalystRunners(artifact.result, {
|
|
3804
|
+
baselineRunnerId: "empty",
|
|
3805
|
+
candidateRunnerId: "model",
|
|
3806
|
+
seed: config.seed
|
|
3807
|
+
})];
|
|
3808
|
+
if (canonicalJson(artifact.comparisons) !== canonicalJson(expectedComparisons)) throw new Error("completed benchmark comparisons do not match durable observations");
|
|
3809
|
+
if (config.dataset === "codetracebench") {
|
|
3810
|
+
if (canonicalJson(artifact.codeTraceCalibration) !== canonicalJson(summarizeCodeTraceCalibration(artifact.result)) || artifact.agentRxCalibration !== void 0) throw new Error("completed CodeTraceBench calibration does not match durable observations");
|
|
3811
|
+
} else if (canonicalJson(artifact.agentRxCalibration) !== canonicalJson(summarizeAgentRxCalibration(artifact.result, "f228165bfec60a801fd5fedd9d8ffe0f9de0c69d")) || artifact.codeTraceCalibration !== void 0) throw new Error("completed AgentRx calibration does not match durable observations");
|
|
3812
|
+
}
|
|
3813
|
+
function assertSameObservations(expected, actual) {
|
|
3814
|
+
if (expected.length !== actual.length) throw new Error(`benchmark result has ${expected.length} observations but the durable log has ${actual.length}`);
|
|
3815
|
+
const expectedByKey = new Map(expected.map((observation) => [observationKey(observation), canonicalJson(observation)]));
|
|
3816
|
+
for (const observation of actual) {
|
|
3817
|
+
const key = observationKey(observation);
|
|
3818
|
+
if (expectedByKey.get(key) !== canonicalJson(observation)) throw new Error(`benchmark result does not match durable observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}'`);
|
|
3819
|
+
expectedByKey.delete(key);
|
|
3820
|
+
}
|
|
3821
|
+
if (expectedByKey.size > 0) throw new Error("benchmark result is missing durable observations");
|
|
3822
|
+
}
|
|
3823
|
+
//#endregion
|
|
3824
|
+
//#region src/analyst/benchmark-report.ts
|
|
3825
|
+
function renderAnalystBenchmarkMarkdown(result, comparisons = []) {
|
|
3826
|
+
const { provenance } = result;
|
|
3827
|
+
const lines = [
|
|
3828
|
+
"# Trace analyst benchmark",
|
|
3829
|
+
"",
|
|
3830
|
+
"## Run",
|
|
3831
|
+
"",
|
|
3832
|
+
"| Field | Value |",
|
|
3833
|
+
"| --- | --- |",
|
|
3834
|
+
`| Benchmark | ${escapeCell(provenance.id ?? "unspecified")} |`,
|
|
3835
|
+
`| Dataset | ${escapeCell(provenance.dataset?.id ?? "unspecified")} |`,
|
|
3836
|
+
`| Dataset revision | ${escapeCell(provenance.dataset?.revision ?? "unspecified")} |`,
|
|
3837
|
+
`| Dataset split | ${escapeCell(provenance.dataset?.split ?? "unspecified")} |`,
|
|
3838
|
+
`| Started | ${escapeCell(provenance.startedAt)} |`,
|
|
3839
|
+
`| Ended | ${escapeCell(provenance.endedAt)} |`,
|
|
3840
|
+
`| Cases | ${provenance.caseCount} |`,
|
|
3841
|
+
`| Runners | ${escapeCell(provenance.runnerIds.join(", "))} |`,
|
|
3842
|
+
`| Repetitions | ${provenance.repetitions} |`,
|
|
3843
|
+
`| Maximum concurrency | ${provenance.maxConcurrency} |`,
|
|
3844
|
+
`| Runner-order seed | ${provenance.runnerOrderSeed} |`,
|
|
3845
|
+
`| Command | ${escapeCell(provenance.command ?? "uncaptured")} |`,
|
|
3846
|
+
`| Environment | ${escapeCell(json(provenance.environment))} |`,
|
|
3847
|
+
`| Metadata | ${escapeCell(json(provenance.metadata))} |`,
|
|
3848
|
+
"",
|
|
3849
|
+
"## Summary",
|
|
3850
|
+
""
|
|
3851
|
+
];
|
|
3852
|
+
lines.push("| Runner | Runs | Failed | Issue-bearing | Trusted negatives | Unlabeled | Micro recall | Micro precision | Micro F1 | Macro recall | Macro precision | Macro F1 | Critical step | Citation coverage | Quote coverage | Label-location agreement | Citation resolution | Resolution unknown runs | Unresolved citations | Resolution errors | Trusted-negative false positives | Trusted-negative failures | Unlabeled prediction rate | Unlabeled failures | Prediction repeat | Prediction repeated cases | Matched-label repeat | Matched-label repeated cases | Latency min/mean/p50/p95/max ms | Locally timed runs | Runner-reported latency runs | Unknown latency | Calls | Input tokens | Output tokens | Reasoning tokens | Cached tokens | Cache-write tokens | Known cost USD | Unknown calls | Unknown input/output | Unknown reasoning | Unknown cached | Unknown cache-write | Unknown cost |", "| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |");
|
|
3853
|
+
for (const summary of result.summaries) lines.push(`| ${escapeCell(summary.runnerId)} | ${summary.completedRuns}/${summary.plannedRuns} | ${summary.failedRuns} | ${summary.issueBearingRuns} | ${summary.trustedNegativeRuns} | ${summary.unlabeledRuns} | ${optionalRate(summary.issueRecall)} | ${optionalRate(summary.findingPrecision)} | ${optionalRate(summary.f1)} | ${optionalRate(summary.macroIssueRecall)} | ${optionalRate(summary.macroFindingPrecision)} | ${optionalRate(summary.macroF1)} | ${optionalRate(summary.criticalStepAccuracy)} | ${optionalRate(summary.citationCoverage)} | ${optionalRate(summary.citationExcerptCoverage)} | ${optionalRate(summary.citationLabelAgreement)} | ${optionalRate(summary.citationResolution)} | ${summary.citationResolutionUnknownRuns} | ${summary.unresolvedCitations} | ${summary.citationResolutionErrors} | ${optionalRate(summary.trustedNegativeFalsePositiveRate)} | ${optionalRate(summary.trustedNegativeFailureRate)} | ${optionalRate(summary.unlabeledPredictionRate)} | ${optionalRate(summary.unlabeledFailureRate)} | ${optionalRate(summary.predictionAgreement)} | ${summary.predictionAgreementCases} | ${optionalRate(summary.matchedLabelAgreement)} | ${summary.matchedLabelAgreementCases} | ${latency(summary.latencyMs)} | ${summary.benchmarkClockLatencyRuns} | ${summary.runnerReportedLatencyRuns} | ${summary.latencyUnknownRuns} | ${summary.calls} | ${summary.inputTokens} | ${summary.outputTokens} | ${summary.reasoningTokens} | ${summary.cachedTokens} | ${summary.cacheWriteTokens} | ${summary.knownCostUsd.toFixed(6)} | ${summary.callsUnknownRuns} | ${summary.tokenUsageUnknownRuns} | ${summary.reasoningTokenUsageUnknownRuns} | ${summary.cachedTokenUsageUnknownRuns} | ${summary.cacheWriteTokenUsageUnknownRuns} | ${summary.costUnknownRuns} |`);
|
|
3854
|
+
for (const comparison of comparisons) {
|
|
3855
|
+
lines.push("", `## ${escapeCell(comparison.candidateRunnerId)} compared with ${escapeCell(comparison.baselineRunnerId)}`, "", "| Metric | Better direction | Paired cases | Independent clusters | Eligible observations | Paired observations | Missing baseline | Missing candidate | Missing asymmetry | Survivor-only | Baseline mean | Candidate mean | Delta | Interval | Minimum sample | Population inference | Limits |", "| --- | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | --- | ---: | ---: | ---: | --- | --- | --- | --- |");
|
|
3856
|
+
for (const metric of comparison.metrics) lines.push(`| ${metric.metric} | ${metric.direction} | ${metric.pairedCases} | ${metric.pairedClusters} | ${metric.eligibleObservations} | ${metric.pairedObservations} | ${metric.baselineMissingObservations} | ${metric.candidateMissingObservations} | ${metric.asymmetricMissingObservations} | ${metric.survivorOnly ? "yes" : "no"} | ${optionalMetricNumber(metric.baselineMean)} | ${optionalMetricNumber(metric.candidateMean)} | ${optionalSigned(metric.meanDelta)} | ${interval(metric.intervalLow, metric.intervalHigh)} | ${metric.minimumSampleMet ? "yes" : "no"} | ${metric.populationInferenceEligible ? "yes" : "no"} | ${escapeCell(metric.inferenceLimitations.join(", ") || "none")} |`);
|
|
3857
|
+
}
|
|
3858
|
+
lines.push("", "## Runs", "", "| Runner | Case | Cluster | Label state | Tags | Case metadata | Runner metadata | Rep | Execution index | Completed | Recall | Precision | F1 | Critical step | Citation coverage | Quote coverage | Label-location agreement | Citation resolution | Prediction on label-empty case | Scored findings | Diagnostic findings | Unlabeled citations | Unresolved citations | Resolution errors | Latency ms | Latency source | Calls | Input tokens | Output tokens | Reasoning tokens | Cached tokens | Cache-write tokens | Cost USD | Known cost USD | Cost source | Error class | Error |", "| --- | --- | --- | --- | --- | --- | --- | ---: | ---: | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | --- | ---: | ---: | ---: | ---: | ---: | ---: | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | --- | --- | --- |");
|
|
3859
|
+
for (const observation of result.observations) {
|
|
3860
|
+
const usage = observation.usage;
|
|
3861
|
+
const cost = usage?.cost.kind === "uncaptured" ? null : usage?.cost.usd;
|
|
3862
|
+
const positive = observation.labelState === "positive";
|
|
3863
|
+
lines.push(`| ${escapeCell(observation.runnerId)} | ${escapeCell(observation.caseId)} | ${escapeCell(observation.clusterId)} | ${observation.labelState} | ${escapeCell(observation.caseTags.join(", "))} | ${escapeCell(json(observation.caseMetadata))} | ${escapeCell(json(observation.runnerMetadata))} | ${observation.repetition} | ${observation.executionIndex} | ${observation.error ? "no" : "yes"} | ${positive ? rate(observation.score.issueRecall) : "n/a"} | ${positive ? rate(observation.score.findingPrecision) : "n/a"} | ${positive ? rate(observation.score.f1) : "n/a"} | ${optionalRate(observation.score.criticalStepAccuracy)} | ${optionalRate(observation.score.citationCoverage)} | ${optionalRate(observation.score.citationExcerptCoverage)} | ${optionalRate(observation.score.citationLabelAgreement)} | ${optionalRate(observation.evidenceResolution?.validity ?? null)} | ${positive ? "n/a" : observation.score.predictionOnLabelEmptyCase ? "yes" : "no"} | ${observation.score.supportedFindingIndexes.length}/${observation.error ? 0 : observation.findings.length} | ${observation.error ? observation.findings.length : 0} | ${observation.score.unlabeledEvidence.length} | ${observation.evidenceResolution?.unresolvedEvidence.length ?? "unknown"} | ${observation.evidenceResolution?.errors.length ?? "unknown"} | ${optionalNumber(observation.latencyMs)} | ${observation.latencySource} | ${usage?.calls ?? "unknown"} | ${usage?.tokens?.input ?? "unknown"} | ${usage?.tokens?.output ?? "unknown"} | ${usage?.tokens?.reasoning ?? "unknown"} | ${usage?.tokens?.cached ?? "unknown"} | ${usage?.tokens?.cacheWrite ?? "unknown"} | ${cost === null || cost === void 0 ? "unknown" : cost.toFixed(6)} | ${usage?.knownCostUsd?.toFixed(6) ?? (cost === null || cost === void 0 ? "unknown" : cost.toFixed(6))} | ${usage?.cost.kind ?? "unknown"} | ${escapeCell(observation.error?.class ?? "")} | ${escapeCell(observation.error?.message ?? "")} |`);
|
|
3864
|
+
}
|
|
3865
|
+
return `${lines.join("\n")}\n`;
|
|
3866
|
+
}
|
|
3867
|
+
function rate(value) {
|
|
3868
|
+
return `${(value * 100).toFixed(1)}%`;
|
|
3869
|
+
}
|
|
3870
|
+
function optionalRate(value) {
|
|
3871
|
+
return value === null ? "n/a" : rate(value);
|
|
3872
|
+
}
|
|
3873
|
+
function number(value) {
|
|
3874
|
+
return Number.isInteger(value) ? String(value) : value.toFixed(3);
|
|
3875
|
+
}
|
|
3876
|
+
function optionalNumber(value) {
|
|
3877
|
+
return value === null ? "unknown" : number(value);
|
|
3878
|
+
}
|
|
3879
|
+
function optionalMetricNumber(value) {
|
|
3880
|
+
return value === null ? "n/a" : number(value);
|
|
3881
|
+
}
|
|
3882
|
+
function signed(value) {
|
|
3883
|
+
return `${value >= 0 ? "+" : ""}${number(value)}`;
|
|
3884
|
+
}
|
|
3885
|
+
function optionalSigned(value) {
|
|
3886
|
+
return value === null ? "n/a" : signed(value);
|
|
3887
|
+
}
|
|
3888
|
+
function interval(low, high) {
|
|
3889
|
+
return low === null || high === null ? "n/a" : `[${number(low)}, ${number(high)}]`;
|
|
3890
|
+
}
|
|
3891
|
+
function latency(value) {
|
|
3892
|
+
if (value === null) return "unknown";
|
|
3893
|
+
return [
|
|
3894
|
+
value.min,
|
|
3895
|
+
value.mean,
|
|
3896
|
+
value.p50,
|
|
3897
|
+
value.p95,
|
|
3898
|
+
value.max
|
|
3899
|
+
].map(number).join("/");
|
|
3900
|
+
}
|
|
3901
|
+
function json(value) {
|
|
3902
|
+
return value === void 0 ? "uncaptured" : JSON.stringify(value);
|
|
3903
|
+
}
|
|
3904
|
+
function escapeCell(value) {
|
|
3905
|
+
return value.replaceAll("|", "\\|").replaceAll("\n", " ");
|
|
3906
|
+
}
|
|
3907
|
+
//#endregion
|
|
3908
|
+
//#region src/analyst/benchmark-command.ts
|
|
3909
|
+
async function runAnalystBenchmarkCommand(argv, env = process.env, dependencies = {}) {
|
|
3910
|
+
if (argv.includes("--help") || argv.includes("-h")) {
|
|
3911
|
+
process.stdout.write(`${ANALYST_BENCHMARK_HELP}\n`);
|
|
3912
|
+
return 0;
|
|
3913
|
+
}
|
|
3914
|
+
const config = parseCommandConfig(argv, env);
|
|
3915
|
+
const outputLock = acquireSingleRunLock({ lockPath: await prepareOutputLockPath(config.outDir) });
|
|
3916
|
+
try {
|
|
3917
|
+
return await executeAnalystBenchmarkCommand(config, dependencies);
|
|
3918
|
+
} finally {
|
|
3919
|
+
outputLock.release();
|
|
3920
|
+
}
|
|
3921
|
+
}
|
|
3922
|
+
async function executeAnalystBenchmarkCommand(config, dependencies) {
|
|
3923
|
+
const paths = await openOutputDirectory(config.outDir, config.resume);
|
|
3924
|
+
const prepared = await preparePublicAnalystBenchmark({
|
|
3925
|
+
dataset: config.dataset,
|
|
3926
|
+
labelsPath: config.labelsPath,
|
|
3927
|
+
traceDir: config.traceDir,
|
|
3928
|
+
artifactDir: config.artifactDir,
|
|
3929
|
+
maxArtifactBytes: config.maxArtifactBytes,
|
|
3930
|
+
limit: config.limit,
|
|
3931
|
+
seed: config.seed
|
|
3932
|
+
});
|
|
3933
|
+
const identity = createRunIdentity(config, prepared);
|
|
3934
|
+
const localReceipt = createLocalRunReceipt(config, paths);
|
|
3935
|
+
const localIdentitySha256 = digestCanonical(localReceipt.local);
|
|
3936
|
+
const identitySha256 = digestCanonical(identity);
|
|
3937
|
+
const manifest = config.resume ? await readAndValidateResumeFiles(paths, identity, identitySha256, localIdentitySha256, localReceipt) : await initializeRunFiles(paths, identity, identitySha256, localIdentitySha256, localReceipt);
|
|
3938
|
+
const progress = await readProgress(paths.observations, manifest.identitySha256, prepared.selectedCaseIds, config.repetitions);
|
|
3939
|
+
const costLedger = createRunCostLedger({
|
|
3940
|
+
storage: fsCampaignStorage(),
|
|
3941
|
+
runDir: paths.directory,
|
|
3942
|
+
costCeilingUsd: config.maxCostUsd
|
|
3943
|
+
});
|
|
3944
|
+
if (await regularFileExists(paths.result)) {
|
|
3945
|
+
assertCostLedgerFinalizable(costLedger);
|
|
3946
|
+
const artifact = await readAnalystBenchmarkArtifact(paths.result);
|
|
3947
|
+
assertCompletedArtifactMatchesRun(artifact, manifest, progress.observations, prepared);
|
|
3948
|
+
const markdown = renderArtifactMarkdown(artifact);
|
|
3949
|
+
await writeExclusiveOrVerify(paths.report, markdown);
|
|
3950
|
+
printSuccessSummary(artifact, paths);
|
|
3951
|
+
return benchmarkExitCode(artifact.result);
|
|
3952
|
+
}
|
|
3953
|
+
if (await regularFileExists(paths.report)) throw new Error(`benchmark report exists without a completed result; refusing ambiguous resume: ${paths.report}`);
|
|
3954
|
+
const createModelRunner = dependencies.createModelRunner ?? ((dataset, model) => createPublicBenchmarkModelRunner(dataset, model));
|
|
3955
|
+
const runners = [emptyPublicBenchmarkRunner(), createModelRunner(config.dataset, {
|
|
3956
|
+
...config.model,
|
|
3957
|
+
costLedger,
|
|
3958
|
+
durability: {
|
|
3959
|
+
runIdentitySha256: manifest.identitySha256,
|
|
3960
|
+
responseCacheDir: paths.modelResponses
|
|
3961
|
+
}
|
|
3962
|
+
})];
|
|
3963
|
+
const appendObservation = createObservationAppender(paths.observations, manifest.identitySha256, progress);
|
|
3964
|
+
const runAbort = new AbortController();
|
|
3965
|
+
let result;
|
|
3966
|
+
try {
|
|
3967
|
+
result = await runAnalystBenchmark({
|
|
3968
|
+
cases: prepared.cases,
|
|
3969
|
+
runners,
|
|
3970
|
+
repetitions: config.repetitions,
|
|
3971
|
+
maxConcurrency: config.concurrency,
|
|
3972
|
+
runnerOrderSeed: config.seed,
|
|
3973
|
+
initialObservations: progress.observations,
|
|
3974
|
+
signal: runAbort.signal,
|
|
3975
|
+
onObservation: async (observation) => {
|
|
3976
|
+
assertObservationAccountingComplete(observation, costLedger);
|
|
3977
|
+
await appendObservation(observation);
|
|
3978
|
+
},
|
|
3979
|
+
resolveEvidence: traceStoreEvidenceResolver((input) => {
|
|
3980
|
+
if (!input.traceStore) throw new Error("benchmark case has no trace store");
|
|
3981
|
+
return input.traceStore;
|
|
3982
|
+
}),
|
|
3983
|
+
benchmark: {
|
|
3984
|
+
id: `${config.dataset}-real-model-analyst`,
|
|
3985
|
+
dataset: {
|
|
3986
|
+
id: config.dataset === "agentrx" ? "microsoft/AgentRx" : "NJU-LINK/CodeTraceBench",
|
|
3987
|
+
revision: config.revision,
|
|
3988
|
+
split: config.split
|
|
3989
|
+
},
|
|
3990
|
+
environment: {
|
|
3991
|
+
node: process.version,
|
|
3992
|
+
platform: platform(),
|
|
3993
|
+
arch: arch()
|
|
3994
|
+
},
|
|
3995
|
+
metadata: {
|
|
3996
|
+
model: config.model.model,
|
|
3997
|
+
outputAdapter: config.dataset === "agentrx" ? "agentrx-taxonomy-and-root-step" : "codetracebench-incorrect-step",
|
|
3998
|
+
caseSelection: prepared.selection.method,
|
|
3999
|
+
caseSelectionSeed: config.seed,
|
|
4000
|
+
selectionStratified: prepared.selection.stratified,
|
|
4001
|
+
protocolSha256: publicBenchmarkProtocolSha256(config.dataset),
|
|
4002
|
+
implementationSha256: ANALYST_BENCHMARK_IMPLEMENTATION_SHA256,
|
|
4003
|
+
dependencyLockSha256: ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256,
|
|
4004
|
+
populationRepresentativenessProven: false
|
|
4005
|
+
}
|
|
4006
|
+
}
|
|
4007
|
+
});
|
|
4008
|
+
} catch (error) {
|
|
4009
|
+
runAbort.abort(error);
|
|
4010
|
+
if (!await costLedger.waitForIdle({ timeoutMs: Math.min(config.model.timeoutMs, 1e4) })) throw accountingError(costLedger, "provider calls remain unresolved after cancellation");
|
|
4011
|
+
throw error;
|
|
4012
|
+
}
|
|
4013
|
+
assertCostLedgerFinalizable(costLedger);
|
|
4014
|
+
result.provenance.startedAt = manifest.createdAt;
|
|
4015
|
+
const persisted = await readProgress(paths.observations, manifest.identitySha256, prepared.selectedCaseIds, config.repetitions);
|
|
4016
|
+
assertSameObservations(result.observations, persisted.observations);
|
|
4017
|
+
const comparisons = [compareAnalystRunners(result, {
|
|
4018
|
+
baselineRunnerId: "empty",
|
|
4019
|
+
candidateRunnerId: "model",
|
|
4020
|
+
seed: config.seed
|
|
4021
|
+
})];
|
|
4022
|
+
const codeTraceCalibration = config.dataset === "codetracebench" ? summarizeCodeTraceCalibration(result) : void 0;
|
|
4023
|
+
const agentRxCalibration = config.dataset === "agentrx" ? summarizeAgentRxCalibration(result, AGENT_RX_UPSTREAM_REVISION) : void 0;
|
|
4024
|
+
const artifact = {
|
|
4025
|
+
kind: "agent-eval/analyst-benchmark-result",
|
|
4026
|
+
runIdentitySha256: manifest.identitySha256,
|
|
4027
|
+
inputs: {
|
|
4028
|
+
dataset: config.dataset,
|
|
4029
|
+
datasetRevision: config.revision,
|
|
4030
|
+
datasetSplit: config.split,
|
|
4031
|
+
labelsSha256: prepared.labelsSha256,
|
|
4032
|
+
sourceRowCount: prepared.sourceRowCount,
|
|
4033
|
+
traceFiles: prepared.traceFiles,
|
|
4034
|
+
verificationArtifacts: prepared.verificationArtifacts,
|
|
4035
|
+
verificationAvailability: summarizeVerificationAvailability(prepared.verificationArtifacts),
|
|
4036
|
+
selection: {
|
|
4037
|
+
limit: config.limit,
|
|
4038
|
+
seed: config.seed,
|
|
4039
|
+
selectedCaseIds: prepared.selectedCaseIds,
|
|
4040
|
+
report: prepared.selection
|
|
4041
|
+
},
|
|
4042
|
+
execution: {
|
|
4043
|
+
repetitions: config.repetitions,
|
|
4044
|
+
concurrency: config.concurrency,
|
|
4045
|
+
model: config.model.model,
|
|
4046
|
+
maxOutputTokens: config.model.maxOutputTokens,
|
|
4047
|
+
timeoutMs: config.model.timeoutMs,
|
|
4048
|
+
maxCostUsd: config.maxCostUsd,
|
|
4049
|
+
maxArtifactBytes: config.maxArtifactBytes,
|
|
4050
|
+
analystProtocolSha256: publicBenchmarkProtocolSha256(config.dataset),
|
|
4051
|
+
implementationSha256: ANALYST_BENCHMARK_IMPLEMENTATION_SHA256,
|
|
4052
|
+
dependencyLockSha256: ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256
|
|
4053
|
+
}
|
|
4054
|
+
},
|
|
4055
|
+
result,
|
|
4056
|
+
comparisons,
|
|
4057
|
+
...codeTraceCalibration ? { codeTraceCalibration } : {},
|
|
4058
|
+
...agentRxCalibration ? { agentRxCalibration } : {}
|
|
4059
|
+
};
|
|
4060
|
+
const markdown = renderArtifactMarkdown(artifact);
|
|
4061
|
+
await writeExclusiveOrVerify(paths.result, `${JSON.stringify(artifact, null, 2)}\n`);
|
|
4062
|
+
await writeExclusiveOrVerify(paths.report, markdown);
|
|
4063
|
+
printSuccessSummary(artifact, paths);
|
|
4064
|
+
return benchmarkExitCode(result);
|
|
4065
|
+
}
|
|
4066
|
+
const NON_SCORABLE_COST_ERRORS = /* @__PURE__ */ new Set([
|
|
4067
|
+
"CostAccountingIncompleteError",
|
|
4068
|
+
"CostCallConflictError",
|
|
4069
|
+
"CostCeilingReachedError",
|
|
4070
|
+
"CostLedgerPersistenceError",
|
|
4071
|
+
"CostReceiptCaptureError",
|
|
4072
|
+
"CostReservationExceededError"
|
|
4073
|
+
]);
|
|
4074
|
+
function assertObservationAccountingComplete(observation, costLedger) {
|
|
4075
|
+
if (observation.error && NON_SCORABLE_COST_ERRORS.has(observation.error.class)) throw new CostAccountingIncompleteError(`Analyst benchmark stopped before scoring: ${observation.error.message}`);
|
|
4076
|
+
if (observation.runnerId !== "model") return;
|
|
4077
|
+
const summary = costLedger.summary({
|
|
4078
|
+
channel: "analyst",
|
|
4079
|
+
tags: {
|
|
4080
|
+
benchmarkCaseId: observation.caseId,
|
|
4081
|
+
benchmarkRepetition: String(observation.repetition)
|
|
4082
|
+
}
|
|
4083
|
+
});
|
|
4084
|
+
if (!summary.accountingComplete || summary.pendingCalls > 0 || summary.unresolvedCalls > 0) throw accountingError(costLedger, "the model observation has incomplete cost accounting", {
|
|
4085
|
+
channel: "analyst",
|
|
4086
|
+
tags: {
|
|
4087
|
+
benchmarkCaseId: observation.caseId,
|
|
4088
|
+
benchmarkRepetition: String(observation.repetition)
|
|
4089
|
+
}
|
|
4090
|
+
});
|
|
4091
|
+
}
|
|
4092
|
+
function assertCostLedgerFinalizable(costLedger) {
|
|
4093
|
+
const summary = costLedger.summary();
|
|
4094
|
+
if (!summary.accountingComplete || summary.pendingCalls > 0 || summary.unresolvedCalls > 0) throw accountingError(costLedger, "the run has pending or incomplete cost entries");
|
|
4095
|
+
}
|
|
4096
|
+
function accountingError(costLedger, reason, filter) {
|
|
4097
|
+
const details = costLedger.summary(filter).incompleteReasons.slice(0, 3).join("; ");
|
|
4098
|
+
return new CostAccountingIncompleteError(`Analyst benchmark cannot continue because ${reason}${details ? `: ${details}` : ""}`);
|
|
4099
|
+
}
|
|
4100
|
+
const ANALYST_BENCHMARK_HELP = `agent-eval analyst-benchmark
|
|
4101
|
+
|
|
4102
|
+
Run a real-model trace analyst against public AgentRx or CodeTraceBench labels.
|
|
4103
|
+
|
|
4104
|
+
Required:
|
|
4105
|
+
--dataset agentrx|codetracebench
|
|
4106
|
+
--labels <dataset.json|dataset.jsonl>
|
|
4107
|
+
--trace-dir <one-trace-per-file OTLP JSONL directory>
|
|
4108
|
+
--artifact-dir <extracted artifact root> Required for CodeTraceBench
|
|
4109
|
+
--out <new output directory>
|
|
4110
|
+
--revision <full 40- or 64-character hex digest>
|
|
4111
|
+
--split <dataset split>
|
|
4112
|
+
--base-url <OpenAI-compatible /v1 URL>
|
|
4113
|
+
--api-key-env <environment variable containing the bearer>
|
|
4114
|
+
--model <provider model id>
|
|
4115
|
+
--limit <positive case count>
|
|
4116
|
+
|
|
4117
|
+
Controls:
|
|
4118
|
+
--resume Continue an interrupted run in --out
|
|
4119
|
+
--seed <integer> Case-selection and comparison seed. Default: 0
|
|
4120
|
+
--concurrency <positive integer> Parallel benchmark jobs. Default: 1
|
|
4121
|
+
--repetitions <positive integer> Runs per case and runner. Default: 1
|
|
4122
|
+
--max-output-tokens <positive> Model output limit per call. Default: 4096
|
|
4123
|
+
--timeout-ms <positive> Model analyst deadline per case. Default: 300000
|
|
4124
|
+
--max-cost-usd <positive> Run-wide spend limit. Default: 5
|
|
4125
|
+
--max-artifact-bytes <positive> Final evidence bytes per case. Default: ${DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES}
|
|
4126
|
+
|
|
4127
|
+
Writes result.json with every observation, metric, usage field, error, comparison,
|
|
4128
|
+
input digest, artifact digest, case distribution, selected case id, and explicit
|
|
4129
|
+
unknown cost. Limited deterministic-hash subsets are marked non-representative.
|
|
4130
|
+
Completed observations are fsynced to observations.jsonl. Shareable output is in
|
|
4131
|
+
result.json and report.md. Machine-local paths, endpoint, and command are isolated
|
|
4132
|
+
in run.local.json.
|
|
4133
|
+
The key is read from the named environment variable and is never written.`;
|
|
4134
|
+
function parseCommandConfig(argv, env) {
|
|
4135
|
+
const flags = parseFlags(argv);
|
|
4136
|
+
assertKnownFlags(flags);
|
|
4137
|
+
const dataset = requiredFlag(flags, "dataset");
|
|
4138
|
+
if (dataset !== "agentrx" && dataset !== "codetracebench") throw new Error("--dataset must be 'agentrx' or 'codetracebench'");
|
|
4139
|
+
const artifactDir = flags.get("artifact-dir")?.trim();
|
|
4140
|
+
if (dataset === "codetracebench" && !artifactDir) throw new Error("--artifact-dir is required for CodeTraceBench");
|
|
4141
|
+
const apiKeyEnv = requiredFlag(flags, "api-key-env");
|
|
4142
|
+
if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(apiKeyEnv)) throw new Error("--api-key-env must be a valid environment variable name");
|
|
4143
|
+
const apiKey = env[apiKeyEnv]?.trim();
|
|
4144
|
+
if (!apiKey) throw new Error(`--api-key-env points to an empty or missing variable: ${apiKeyEnv}`);
|
|
4145
|
+
return {
|
|
4146
|
+
dataset,
|
|
4147
|
+
labelsPath: requiredFlag(flags, "labels"),
|
|
4148
|
+
traceDir: requiredFlag(flags, "trace-dir"),
|
|
4149
|
+
...artifactDir ? { artifactDir } : {},
|
|
4150
|
+
outDir: requiredFlag(flags, "out"),
|
|
4151
|
+
revision: immutableRevision(requiredFlag(flags, "revision")),
|
|
4152
|
+
split: requiredFlag(flags, "split"),
|
|
4153
|
+
model: {
|
|
4154
|
+
baseUrl: openAiCompatibleBaseUrl(requiredFlag(flags, "base-url")),
|
|
4155
|
+
apiKey,
|
|
4156
|
+
model: requiredFlag(flags, "model"),
|
|
4157
|
+
maxOutputTokens: positiveFlag(flags, "max-output-tokens", 4096),
|
|
4158
|
+
timeoutMs: positiveFlag(flags, "timeout-ms", 3e5)
|
|
4159
|
+
},
|
|
4160
|
+
limit: positiveFlag(flags, "limit"),
|
|
4161
|
+
seed: integerFlag(flags, "seed", 0),
|
|
4162
|
+
concurrency: positiveFlag(flags, "concurrency", 1),
|
|
4163
|
+
repetitions: positiveFlag(flags, "repetitions", 1),
|
|
4164
|
+
maxCostUsd: positiveFiniteFlag(flags, "max-cost-usd", 5),
|
|
4165
|
+
maxArtifactBytes: positiveFlag(flags, "max-artifact-bytes", DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES),
|
|
4166
|
+
apiKeyEnv,
|
|
4167
|
+
command: `agent-eval analyst-benchmark ${argv.filter((argument) => argument !== "--resume").map(shellQuote).join(" ")}`,
|
|
4168
|
+
resume: flags.has("resume")
|
|
4169
|
+
};
|
|
4170
|
+
}
|
|
4171
|
+
function parseFlags(argv) {
|
|
4172
|
+
const flags = /* @__PURE__ */ new Map();
|
|
4173
|
+
for (let index = 0; index < argv.length; index += 1) {
|
|
4174
|
+
const token = argv[index];
|
|
4175
|
+
if (!token.startsWith("--")) throw new Error(`unexpected positional argument: ${token}`);
|
|
4176
|
+
const raw = token.slice(2);
|
|
4177
|
+
const equalsAt = raw.indexOf("=");
|
|
4178
|
+
const name = equalsAt < 0 ? raw : raw.slice(0, equalsAt);
|
|
4179
|
+
const inlineValue = equalsAt < 0 ? void 0 : raw.slice(equalsAt + 1);
|
|
4180
|
+
if (!name || flags.has(name)) throw new Error(`duplicate or empty flag: --${name}`);
|
|
4181
|
+
if (BOOLEAN_FLAGS.has(name)) {
|
|
4182
|
+
if (inlineValue !== void 0) throw new Error(`--${name} does not accept a value`);
|
|
4183
|
+
flags.set(name, "true");
|
|
4184
|
+
continue;
|
|
4185
|
+
}
|
|
4186
|
+
const value = inlineValue ?? argv[++index];
|
|
4187
|
+
if (!value || value.startsWith("--")) throw new Error(`--${name} requires a value`);
|
|
4188
|
+
flags.set(name, value);
|
|
4189
|
+
}
|
|
4190
|
+
return flags;
|
|
4191
|
+
}
|
|
4192
|
+
const KNOWN_FLAGS = /* @__PURE__ */ new Set([
|
|
4193
|
+
"resume",
|
|
4194
|
+
"dataset",
|
|
4195
|
+
"labels",
|
|
4196
|
+
"trace-dir",
|
|
4197
|
+
"artifact-dir",
|
|
4198
|
+
"out",
|
|
4199
|
+
"revision",
|
|
4200
|
+
"split",
|
|
4201
|
+
"base-url",
|
|
4202
|
+
"api-key-env",
|
|
4203
|
+
"model",
|
|
4204
|
+
"limit",
|
|
4205
|
+
"seed",
|
|
4206
|
+
"concurrency",
|
|
4207
|
+
"repetitions",
|
|
4208
|
+
"max-output-tokens",
|
|
4209
|
+
"timeout-ms",
|
|
4210
|
+
"max-cost-usd",
|
|
4211
|
+
"max-artifact-bytes"
|
|
4212
|
+
]);
|
|
4213
|
+
const BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["resume"]);
|
|
4214
|
+
function assertKnownFlags(flags) {
|
|
4215
|
+
for (const flag of flags.keys()) if (!KNOWN_FLAGS.has(flag)) throw new Error(`unknown analyst-benchmark flag: --${flag}`);
|
|
4216
|
+
}
|
|
4217
|
+
function requiredFlag(flags, name) {
|
|
4218
|
+
const value = flags.get(name)?.trim();
|
|
4219
|
+
if (!value) throw new Error(`--${name} is required`);
|
|
4220
|
+
return value;
|
|
4221
|
+
}
|
|
4222
|
+
function positiveFlag(flags, name, defaultValue) {
|
|
4223
|
+
const raw = flags.get(name);
|
|
4224
|
+
if (raw === void 0 && defaultValue !== void 0) return defaultValue;
|
|
4225
|
+
if (raw === void 0) throw new Error(`--${name} is required`);
|
|
4226
|
+
const value = Number(raw);
|
|
4227
|
+
if (!Number.isSafeInteger(value) || value < 1) throw new Error(`--${name} must be a positive safe integer`);
|
|
4228
|
+
return value;
|
|
4229
|
+
}
|
|
4230
|
+
function integerFlag(flags, name, defaultValue) {
|
|
4231
|
+
const raw = flags.get(name);
|
|
4232
|
+
if (raw === void 0) return defaultValue;
|
|
4233
|
+
const value = Number(raw);
|
|
4234
|
+
if (!Number.isSafeInteger(value)) throw new Error(`--${name} must be a safe integer`);
|
|
4235
|
+
return value;
|
|
4236
|
+
}
|
|
4237
|
+
function positiveFiniteFlag(flags, name, defaultValue) {
|
|
4238
|
+
const raw = flags.get(name);
|
|
4239
|
+
if (raw === void 0) return defaultValue;
|
|
4240
|
+
const value = Number(raw);
|
|
4241
|
+
if (!Number.isFinite(value) || value <= 0) throw new Error(`--${name} must be a positive finite number`);
|
|
4242
|
+
return value;
|
|
4243
|
+
}
|
|
4244
|
+
function openAiCompatibleBaseUrl(value) {
|
|
4245
|
+
let parsed;
|
|
4246
|
+
try {
|
|
4247
|
+
parsed = new URL(value);
|
|
4248
|
+
} catch {
|
|
4249
|
+
throw new Error("--base-url must be an absolute HTTP or HTTPS URL");
|
|
4250
|
+
}
|
|
4251
|
+
if (parsed.protocol !== "http:" && parsed.protocol !== "https:" || parsed.username || parsed.password || parsed.search || parsed.hash) throw new Error("--base-url must use HTTP or HTTPS without credentials, query, or fragment");
|
|
4252
|
+
if (parsed.protocol === "http:" && !isLoopbackHost(parsed.hostname)) throw new Error("--base-url must use HTTPS unless the endpoint is on the local machine");
|
|
4253
|
+
return value;
|
|
4254
|
+
}
|
|
4255
|
+
function isLoopbackHost(hostname) {
|
|
4256
|
+
const normalized = hostname.toLowerCase();
|
|
4257
|
+
return normalized === "localhost" || normalized === "[::1]" || normalized === "::1" || /^127(?:\.\d{1,3}){3}$/.test(normalized);
|
|
4258
|
+
}
|
|
4259
|
+
function immutableRevision(value) {
|
|
4260
|
+
if (!/^(?:[a-fA-F0-9]{40}|[a-fA-F0-9]{64})$/.test(value)) throw new Error("--revision must be a full 40- or 64-character hexadecimal digest");
|
|
4261
|
+
return value.toLowerCase();
|
|
4262
|
+
}
|
|
4263
|
+
function renderSelectionMarkdown(report) {
|
|
4264
|
+
const rows = [
|
|
4265
|
+
"class",
|
|
4266
|
+
"agent",
|
|
4267
|
+
"model",
|
|
4268
|
+
"difficulty",
|
|
4269
|
+
"solved"
|
|
4270
|
+
].map((dimension) => {
|
|
4271
|
+
const source = report.source[dimension];
|
|
4272
|
+
const selected = report.selected[dimension];
|
|
4273
|
+
return `| ${dimension} | ${distributionText(source.counts, source.missing, source.total)} | ${distributionText(selected.counts, selected.missing, selected.total)} |`;
|
|
4274
|
+
});
|
|
4275
|
+
return [
|
|
4276
|
+
"## Case Selection",
|
|
4277
|
+
"",
|
|
4278
|
+
`Method: \`${report.method}\`; seed: \`${report.seed}\`; selected: ${report.selectedCount}/${report.sourceCount}.`,
|
|
4279
|
+
report.representativeOfInput ? "This is a census of the supplied input." : "This deterministic hash subset is not stratified and must not be presented as representative.",
|
|
4280
|
+
"",
|
|
4281
|
+
"| Dimension | Supplied input | Selected cases |",
|
|
4282
|
+
"| --- | --- | --- |",
|
|
4283
|
+
...rows
|
|
4284
|
+
].join("\n");
|
|
4285
|
+
}
|
|
4286
|
+
function distributionText(counts, missing, total) {
|
|
4287
|
+
const values = Object.entries(counts).map(([value, count]) => `${value}=${count}`);
|
|
4288
|
+
if (missing > 0) values.push(`missing=${missing}`);
|
|
4289
|
+
return `${values.join(", ") || "none"} (n=${total})`;
|
|
4290
|
+
}
|
|
4291
|
+
function summarizeVerificationAvailability(manifests) {
|
|
4292
|
+
return {
|
|
4293
|
+
cases: manifests.length,
|
|
4294
|
+
resultFilesPresent: manifests.filter((manifest) => manifest.status === "present").length,
|
|
4295
|
+
resultFilesMissing: manifests.filter((manifest) => manifest.status === "missing").length,
|
|
4296
|
+
outcomes: {
|
|
4297
|
+
passed: manifests.filter((manifest) => manifest.outcome.status === "passed").length,
|
|
4298
|
+
failed: manifests.filter((manifest) => manifest.outcome.status === "failed").length,
|
|
4299
|
+
unavailable: manifests.filter((manifest) => manifest.outcome.status === "unavailable").length
|
|
4300
|
+
}
|
|
4301
|
+
};
|
|
4302
|
+
}
|
|
4303
|
+
function renderVerificationAvailability(summary) {
|
|
4304
|
+
return [
|
|
4305
|
+
"## Final Verification Availability",
|
|
4306
|
+
"",
|
|
4307
|
+
"| Cases | Result files present | Result files missing | Passed | Failed | Unavailable |",
|
|
4308
|
+
"| ---: | ---: | ---: | ---: | ---: | ---: |",
|
|
4309
|
+
`| ${summary.cases} | ${summary.resultFilesPresent} | ${summary.resultFilesMissing} | ${summary.outcomes.passed} | ${summary.outcomes.failed} | ${summary.outcomes.unavailable} |`
|
|
4310
|
+
].join("\n");
|
|
4311
|
+
}
|
|
4312
|
+
function renderArtifactMarkdown(artifact) {
|
|
4313
|
+
const calibrationMarkdown = artifact.codeTraceCalibration ? `\n\n${renderCodeTraceCalibrationMarkdown(artifact.codeTraceCalibration)}` : artifact.agentRxCalibration ? `\n\n${renderAgentRxCalibrationMarkdown(artifact.agentRxCalibration)}` : "";
|
|
4314
|
+
const verificationMarkdown = artifact.inputs.dataset === "codetracebench" ? `\n\n${renderVerificationAvailability(artifact.inputs.verificationAvailability)}` : "";
|
|
4315
|
+
return `${renderAnalystBenchmarkMarkdown(artifact.result, artifact.comparisons).trimEnd()}${calibrationMarkdown}${verificationMarkdown}\n\n${renderSelectionMarkdown(artifact.inputs.selection.report)}\n`;
|
|
4316
|
+
}
|
|
4317
|
+
function benchmarkExitCode(result) {
|
|
4318
|
+
return result.summaries.find((summary) => summary.runnerId === "model")?.failedRuns ? 2 : 0;
|
|
4319
|
+
}
|
|
4320
|
+
function printSuccessSummary(artifact, paths) {
|
|
4321
|
+
const failures = artifact.result.summaries.reduce((total, summary) => total + summary.failedRuns, 0);
|
|
4322
|
+
const knownCostUsd = artifact.result.summaries.reduce((total, summary) => total + summary.knownCostUsd, 0);
|
|
4323
|
+
const unknownCostRuns = artifact.result.summaries.reduce((total, summary) => total + summary.costUnknownRuns, 0);
|
|
4324
|
+
process.stdout.write(`Analyst benchmark complete: cases=${artifact.result.provenance.caseCount} failures=${failures} known_cost_usd=${knownCostUsd.toFixed(6)} unknown_cost_runs=${unknownCostRuns}\nresult=${paths.result}\nreport=${paths.report}\n`);
|
|
4325
|
+
}
|
|
4326
|
+
function shellQuote(value) {
|
|
4327
|
+
return /^[a-zA-Z0-9_./:@%+=,-]+$/.test(value) ? value : `'${value.replaceAll("'", "'\\''")}'`;
|
|
4328
|
+
}
|
|
4329
|
+
//#endregion
|
|
4330
|
+
export { analystBenchmarkDependencyLockDigest as A, codeTracerPredictionsToFindings as B, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as C, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as D, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as E, ANALYST_BENCHMARK_OBSERVATIONS_FILE as F, normalizeBenchmarkLabel as G, agentRxPredictionsToFindings as H, AGENT_RX_UPSTREAM_REVISION as I, renderAgentRxCalibrationMarkdown as L, ANALYST_BENCHMARK_COST_LEDGER_FILE as M, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as N, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as O, ANALYST_BENCHMARK_MANIFEST_FILE as P, summarizeAgentRxCalibration as R, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as S, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as T, normalizeAgentRxCategory as U, agentRxBenchmarkCase as V, roundAgentRxStep as W, appendVerificationArtifactsToOtlp as _, renderCodeTraceCalibrationMarkdown as a, adaptPublicBenchmarkFindings as b, CODE_TRACE_BENCH_ANALYST_PROMPT as c, loadPublicBenchmarkRows as d, preparePublicAnalystBenchmark as f, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as g, selectPublicBenchmarkRows as h, readAnalystBenchmarkArtifact as i, analystBenchmarkImplementationDigest as j, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as k, createPublicBenchmarkModelRunner as l, publicBenchmarkSelectionReport as m, runAnalystBenchmarkCommand as n, summarizeCodeTraceCalibration as o, publicBenchmarkDistributions as p, renderAnalystBenchmarkMarkdown as r, compareAnalystRunners as s, ANALYST_BENCHMARK_HELP as t, publicBenchmarkProtocolSha256 as u, loadCodeTraceVerificationArtifacts as v, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as w, emptyPublicBenchmarkRunner as x, parseVerificationOutcome as y, codeTraceBenchCase as z };
|
|
4331
|
+
|
|
4332
|
+
//# sourceMappingURL=benchmark-command-CMqVqReF.js.map
|