@tangle-network/agent-bench 0.11.1 → 0.11.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,15 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.11.3
|
|
4
|
+
|
|
5
|
+
Uses Runtime 0.209.0, Eval 0.180.0, and Knowledge 15.0.3 together, including named resource accounting and source-attributed resource reports.
|
|
6
|
+
SWE-bench setup, prompts, and patch extraction use the writable session workspace instead of a root-level directory.
|
|
7
|
+
Official grading behavior is unchanged.
|
|
8
|
+
|
|
9
|
+
## 0.11.2
|
|
10
|
+
|
|
11
|
+
Supports SWE-bench 5.x by detecting removed cache and namespace evaluator flags while retaining the 4.x path.
|
|
12
|
+
|
|
3
13
|
## 0.11.1
|
|
4
14
|
|
|
5
15
|
The published benchmark package now admits sandbox SDK 0.38.x, including consumer-ready runtime-edge readiness.
|
|
@@ -14,13 +14,8 @@ import { join } from "node:path";
|
|
|
14
14
|
* SWE-specific pieces: the patch OutputAdapter, the dataset dump, and the
|
|
15
15
|
* predictions-file → run_evaluation argv → report-shape mapping.
|
|
16
16
|
*/
|
|
17
|
-
/**
|
|
18
|
-
|
|
19
|
-
* source of truth shared by the prompt template (which tells the agent to clone
|
|
20
|
-
* here) and `boxExtract` (which runs `git diff` here after the shot) — so the
|
|
21
|
-
* harness always knows exactly where the agent's edits live, for any instance.
|
|
22
|
-
*/
|
|
23
|
-
const SWE_REPO_DIR = "/work";
|
|
17
|
+
/** Root-level directories are not writable in every sandbox; use the session workspace. */
|
|
18
|
+
const SWE_REPO_DIR = "./swe-bench-repo";
|
|
24
19
|
/**
|
|
25
20
|
* The SWE deliverable's FALLBACK parser, from the agent's event STREAM.
|
|
26
21
|
*
|
|
@@ -97,7 +92,10 @@ function scoreSweReport(taskId, value) {
|
|
|
97
92
|
};
|
|
98
93
|
}
|
|
99
94
|
function sweEvaluationArgv(args) {
|
|
100
|
-
return
|
|
95
|
+
return sweEvaluationArgvForHarness(args, true);
|
|
96
|
+
}
|
|
97
|
+
function sweEvaluationArgvForHarness(args, legacyFlags) {
|
|
98
|
+
const common = [
|
|
101
99
|
"-m",
|
|
102
100
|
"swebench.harness.run_evaluation",
|
|
103
101
|
"--dataset_name",
|
|
@@ -109,7 +107,11 @@ function sweEvaluationArgv(args) {
|
|
|
109
107
|
"--instance_ids",
|
|
110
108
|
args.instanceId,
|
|
111
109
|
"--max_workers",
|
|
112
|
-
"1"
|
|
110
|
+
"1"
|
|
111
|
+
];
|
|
112
|
+
if (!legacyFlags) return common;
|
|
113
|
+
return [
|
|
114
|
+
...common,
|
|
113
115
|
"--namespace",
|
|
114
116
|
args.namespace ?? scorerNamespace(),
|
|
115
117
|
"--cache_level",
|
|
@@ -135,6 +137,7 @@ function createSweBenchAdapter(options = {}) {
|
|
|
135
137
|
if (!SWE_CACHE_LEVELS.has(cacheLevel)) throw new Error("swe-bench: invalid cacheLevel");
|
|
136
138
|
if (options.captureEvaluatorArtifacts !== void 0 && typeof options.captureEvaluatorArtifacts !== "function") throw new Error("swe-bench: captureEvaluatorArtifacts must be a function");
|
|
137
139
|
let attemptSequence = 0;
|
|
140
|
+
let legacyFlags;
|
|
138
141
|
return {
|
|
139
142
|
name: "swe-bench-verified",
|
|
140
143
|
output: swePatchOutput,
|
|
@@ -149,8 +152,9 @@ function createSweBenchAdapter(options = {}) {
|
|
|
149
152
|
await preflightVenvImports({
|
|
150
153
|
modules: ["swebench"],
|
|
151
154
|
requireDocker: true,
|
|
152
|
-
fix: "Fix: (1) python3 -m venv bench/.venv && bench/.venv/bin/pip install swebench ; (2) ensure the Docker daemon is running (the judge builds per-instance images)."
|
|
155
|
+
fix: "Fix: (1) python3 -m venv bench/.venv && bench/.venv/bin/pip install 'swebench>=4,<6' ; (2) ensure the Docker daemon is running (the judge builds per-instance images)."
|
|
153
156
|
});
|
|
157
|
+
legacyFlags = await detectLegacyFlags();
|
|
154
158
|
},
|
|
155
159
|
async loadTasks(opts = {}) {
|
|
156
160
|
const limit = opts.limit ?? 10;
|
|
@@ -180,10 +184,10 @@ print(json.dumps(out))
|
|
|
180
184
|
prompt: [
|
|
181
185
|
`Repository: ${r.repo} @ ${r.base_commit}`,
|
|
182
186
|
"",
|
|
183
|
-
`The repository is ALREADY cloned at ${SWE_REPO_DIR}, checked out at commit ${r.base_commit}. Work there directly (\`cd ${SWE_REPO_DIR}\`); do not re-clone.`,
|
|
187
|
+
`The repository is ALREADY cloned at ${SWE_REPO_DIR} relative to your initial session workspace, checked out at commit ${r.base_commit}. Work there directly (\`cd ${SWE_REPO_DIR}\`); do not re-clone.`,
|
|
184
188
|
"",
|
|
185
189
|
"Resolve this issue by editing the repository SOURCE so the failing tests pass without breaking the passing ones. Do NOT edit test files — the evaluation runs hidden tests on a fresh checkout, so editing tests does not count. Keep the change minimal and confined to the cloned repo.",
|
|
186
|
-
"Work iteratively: reproduce the issue, implement the fix in the source, and re-run the relevant tests until they pass. You do NOT need to print the diff — the harness reads your
|
|
190
|
+
"Work iteratively: reproduce the issue, implement the fix in the source, and re-run the relevant tests until they pass. You do NOT need to print the diff — the harness reads your source edits directly from the repo, including uncommitted changes.",
|
|
187
191
|
"",
|
|
188
192
|
"--- Issue ---",
|
|
189
193
|
String(r.problem_statement ?? "")
|
|
@@ -196,6 +200,7 @@ print(json.dumps(out))
|
|
|
196
200
|
return typeof gold === "string" ? gold : void 0;
|
|
197
201
|
},
|
|
198
202
|
async judge(task, artifact) {
|
|
203
|
+
const useLegacyFlags = await ensureLegacyFlags();
|
|
199
204
|
const runId = safeRunId("bench", task.id);
|
|
200
205
|
const capture = options.captureEvaluatorArtifacts?.({
|
|
201
206
|
taskId: task.id,
|
|
@@ -214,13 +219,13 @@ print(json.dumps(out))
|
|
|
214
219
|
model_patch: artifact
|
|
215
220
|
}]));
|
|
216
221
|
},
|
|
217
|
-
argv: (dir) =>
|
|
222
|
+
argv: (dir) => sweEvaluationArgvForHarness({
|
|
218
223
|
predictionsPath: join(dir, "preds.json"),
|
|
219
224
|
runId,
|
|
220
225
|
instanceId: task.id,
|
|
221
226
|
cacheLevel,
|
|
222
227
|
namespace: scorerNamespace()
|
|
223
|
-
}),
|
|
228
|
+
}, useLegacyFlags),
|
|
224
229
|
async parseReport(dir) {
|
|
225
230
|
const report = await readJsonReport(join(dir, `agent-runtime-bench.${runId}.json`));
|
|
226
231
|
return scoreSweReport(task.id, report);
|
|
@@ -228,6 +233,19 @@ print(json.dumps(out))
|
|
|
228
233
|
});
|
|
229
234
|
}
|
|
230
235
|
};
|
|
236
|
+
async function detectLegacyFlags() {
|
|
237
|
+
return (await runVenvPython(`
|
|
238
|
+
import subprocess, sys
|
|
239
|
+
result = subprocess.run([sys.executable, '-m', 'swebench.harness.run_evaluation', '--help'], capture_output=True, text=True)
|
|
240
|
+
if result.returncode != 0:
|
|
241
|
+
raise SystemExit(result.stderr or result.stdout or str(result.returncode))
|
|
242
|
+
print('legacy' if '--namespace' in result.stdout and '--cache_level' in result.stdout else 'modern')
|
|
243
|
+
`)).trim() === "legacy";
|
|
244
|
+
}
|
|
245
|
+
async function ensureLegacyFlags() {
|
|
246
|
+
if (legacyFlags === void 0) legacyFlags = await detectLegacyFlags();
|
|
247
|
+
return legacyFlags;
|
|
248
|
+
}
|
|
231
249
|
}
|
|
232
250
|
//#endregion
|
|
233
251
|
export { createSweBenchAdapter, scoreSweReport, sweEvaluationArgv, swePatchOutput };
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"swe-bench.js","names":[],"sources":["../../src/benchmarks/swe-bench.ts"],"sourcesContent":["/**\n * SWE-bench Verified adapter. Worker artifact = a unified-diff patch. Judge =\n * the official `swebench` harness: apply the patch in the instance's Docker\n * image, run FAIL_TO_PASS + PASS_TO_PASS, report `resolved`. Fully deterministic\n * — no LLM judge.\n *\n * Requires: the bench `.venv` with `swebench` installed + a running Docker\n * daemon (per-instance images are pulled/built on first run).\n *\n * Process/Docker/report plumbing is shared via ./_harness; this file owns the\n * SWE-specific pieces: the patch OutputAdapter, the dataset dump, and the\n * predictions-file → run_evaluation argv → report-shape mapping.\n */\n\nimport { join } from 'node:path'\nimport type { OutputAdapter } from '@tangle-network/agent-runtime/kernel'\nimport {\n preflightVenvImports,\n readJsonReport,\n runStagedJudge,\n runVenvPython,\n safeRunId,\n stageFile,\n type StagedRunCaptureSpec,\n} from './_harness'\nimport type { BenchmarkAdapter, BenchScore, BenchTask, LoadOptions } from './types'\n\n/**\n * Fixed in-box path the agent clones the instance repo into. It is the SINGLE\n * source of truth shared by the prompt template (which tells the agent to clone\n * here) and `boxExtract` (which runs `git diff` here after the shot) — so the\n * harness always knows exactly where the agent's edits live, for any instance.\n */\nconst SWE_REPO_DIR = '/work'\n\n/**\n * The SWE deliverable's FALLBACK parser, from the agent's event STREAM.\n *\n * The PRIMARY deliverable is `boxExtract` below: a `git diff` of the agent's\n * actual edits, read from the cloned repo's STATE inside the box (standard\n * SWE-bench practice). This event-stream parse only runs when that diff is empty\n * — a model that edited the source correctly but never printed a fenced diff (the\n * exact failure this replaces) still scores off its real changes, not its prose.\n */\nexport const swePatchOutput: OutputAdapter<string> = {\n parse(events) {\n let text = ''\n for (const ev of events) {\n const d = (ev as { data?: Record<string, unknown> })?.data\n const t = d?.finalText ?? d?.text ?? d?.result\n if (typeof t === 'string' && t.length > 0) text = t\n }\n // Last ```diff/```patch fenced block (the contract the prompt asks for);\n // fall back to the raw text so a fence-less but valid diff still reaches the judge.\n const fences = [...text.matchAll(/```(?:diff|patch)?\\s*\\n([\\s\\S]*?)```/g)]\n const last = fences.at(-1)?.[1]\n return (last ?? text).trim()\n },\n}\n\nconst DATASET = 'princeton-nlp/SWE-bench_Verified'\nexport type SweBenchCacheLevel = 'none' | 'base' | 'env' | 'instance'\n\nexport interface SweBenchArtifactCaptureContext {\n readonly taskId: string\n readonly runId: string\n /** One-based sequence unique within this adapter instance. */\n readonly attemptSequence: number\n}\n\nexport interface SweBenchAdapterOptions {\n readonly timeoutMs?: number\n readonly cacheLevel?: SweBenchCacheLevel\n /**\n * Return a unique destination for any attempt whose complete official\n * evaluator directory and process logs should be retained.\n */\n readonly captureEvaluatorArtifacts?: (\n context: SweBenchArtifactCaptureContext,\n ) => StagedRunCaptureSpec | undefined\n}\n\nconst SWE_CACHE_LEVELS = new Set<SweBenchCacheLevel>(['none', 'base', 'env', 'instance'])\n\nfunction scorerNamespace(): 'swebench' | 'none' {\n const namespace = process.env.SWEBENCH_NAMESPACE ?? 'swebench'\n if (namespace !== 'swebench' && namespace !== 'none') {\n throw new Error(`SWEBENCH_NAMESPACE must be swebench|none, got \"${namespace}\"`)\n }\n return namespace\n}\nconst TEST_FILE_EXCLUDES = [\n \"':(exclude,glob)**/tests/**'\",\n \"':(exclude,glob)**/test/**'\",\n \"':(exclude,glob)test_*.py'\",\n \"':(exclude,glob)**/test_*.py'\",\n \"':(exclude,glob)*_test.py'\",\n \"':(exclude,glob)**/*_test.py'\",\n \"':(exclude,glob)conftest.py'\",\n \"':(exclude,glob)**/conftest.py'\",\n].join(' ')\n\ninterface SweReport {\n resolved_instances?: number\n resolved_ids?: string[]\n unresolved_ids?: string[]\n empty_patch_ids?: string[]\n completed_ids?: string[]\n incomplete_ids?: string[]\n error_ids?: string[]\n submitted_ids?: string[]\n}\n\nfunction stringIds(report: Record<string, unknown>, key: keyof SweReport): string[] {\n const value = report[key]\n if (value === undefined) return []\n if (!Array.isArray(value) || value.some((entry) => typeof entry !== 'string')) {\n throw new Error(`swe-bench: malformed ${key}`)\n }\n return value\n}\n\n/** Convert one official report into a score without turning evaluator failures into agent failures. */\nexport function scoreSweReport(taskId: string, value: unknown): BenchScore {\n if (!value || typeof value !== 'object' || Array.isArray(value)) {\n throw new Error('swe-bench: report must be an object')\n }\n const report = value as Record<string, unknown>\n const statusIds = {\n resolved: stringIds(report, 'resolved_ids'),\n unresolved: stringIds(report, 'unresolved_ids'),\n emptyPatch: stringIds(report, 'empty_patch_ids'),\n completed: stringIds(report, 'completed_ids'),\n incomplete: stringIds(report, 'incomplete_ids'),\n error: stringIds(report, 'error_ids'),\n }\n const submitted = stringIds(report, 'submitted_ids')\n const mentioned = Object.values(statusIds).flat()\n if (\n mentioned.some((id) => id !== taskId)\n || (submitted.length > 0 && (submitted.length !== 1 || submitted[0] !== taskId))\n ) {\n throw new Error(`swe-bench: report identity mismatch for ${taskId}`)\n }\n if (statusIds.error.includes(taskId) || statusIds.incomplete.includes(taskId)) {\n throw new Error(`swe-bench: evaluator failed for ${taskId}`)\n }\n const outcomes = [\n statusIds.resolved.includes(taskId),\n statusIds.unresolved.includes(taskId),\n statusIds.emptyPatch.includes(taskId),\n ]\n if (outcomes.filter(Boolean).length !== 1) {\n throw new Error(`swe-bench: report has no unique outcome for ${taskId}`)\n }\n if ((outcomes[0] || outcomes[1]) && !statusIds.completed.includes(taskId)) {\n throw new Error(`swe-bench: report lacks a completed evaluation for ${taskId}`)\n }\n const resolved = outcomes[0]\n return { resolved, score: resolved ? 1 : 0, detail: JSON.stringify(report) }\n}\n\nexport function sweEvaluationArgv(args: {\n readonly predictionsPath: string\n readonly runId: string\n readonly instanceId: string\n readonly cacheLevel: SweBenchCacheLevel\n readonly namespace?: 'swebench' | 'none'\n}): string[] {\n return [\n '-m', 'swebench.harness.run_evaluation',\n '--dataset_name', DATASET,\n '--predictions_path', args.predictionsPath,\n '--run_id', args.runId,\n '--instance_ids', args.instanceId,\n '--max_workers', '1',\n '--namespace', args.namespace ?? scorerNamespace(),\n '--cache_level', args.cacheLevel,\n ]\n}\n\nfunction shellQuote(value: string): string {\n return `'${value.replace(/'/g, `'\\\\''`)}'`\n}\n\nfunction sweMetadata(task: BenchTask): { repo: string; base: string } {\n const repo = String(task.metadata?.repo ?? '')\n const base = String(task.metadata?.base_commit ?? '')\n if (!/^[A-Za-z0-9_.-]+\\/[A-Za-z0-9_.-]+$/.test(repo)) {\n throw new Error(`swe-bench: invalid repo metadata for ${task.id}: ${repo}`)\n }\n if (!/^[0-9a-f]{7,40}$/i.test(base)) {\n throw new Error(`swe-bench: invalid base_commit metadata for ${task.id}: ${base}`)\n }\n return { repo, base }\n}\n\nexport function createSweBenchAdapter(options: SweBenchAdapterOptions = {}): BenchmarkAdapter {\n if (\n options.timeoutMs !== undefined\n && (!Number.isSafeInteger(options.timeoutMs) || options.timeoutMs <= 0)\n ) throw new Error('swe-bench: timeoutMs must be a positive integer')\n const cacheLevel = options.cacheLevel ?? 'env'\n if (!SWE_CACHE_LEVELS.has(cacheLevel)) throw new Error('swe-bench: invalid cacheLevel')\n if (\n options.captureEvaluatorArtifacts !== undefined\n && typeof options.captureEvaluatorArtifacts !== 'function'\n ) throw new Error('swe-bench: captureEvaluatorArtifacts must be a function')\n let attemptSequence = 0\n return {\n name: 'swe-bench-verified',\n output: swePatchOutput,\n\n // Extract the patch from repo STATE, not printed text: stage every edit the\n // agent made in the cloned repo and diff it against the checked-out\n // base_commit (`HEAD`). Test files are excluded — the judge applies the gold\n // `test_patch` itself, so an agent edit to a test would collide on apply. The\n // paths come out `a/<repo-relative>` (cwd = repo root), matching the gold\n // patch format the swebench judge's `git apply` expects.\n // Pre-stage: clone the instance repo at base_commit into SWE_REPO_DIR so the\n // agent only edits (the harness owns the checkout — a stochastic model can't be\n // trusted to clone to an exact path). `--quiet` keeps the exec output small.\n boxSetup(task) {\n const { repo, base } = sweMetadata(task)\n return {\n command: `rm -rf ${shellQuote(SWE_REPO_DIR)} && git clone --quiet ${shellQuote(`https://github.com/${repo}`)} ${shellQuote(SWE_REPO_DIR)} && git -C ${shellQuote(SWE_REPO_DIR)} checkout --quiet ${shellQuote(base)}`,\n }\n },\n boxExtract() {\n return {\n command: `git -C ${shellQuote(SWE_REPO_DIR)} add -A && git -C ${shellQuote(SWE_REPO_DIR)} diff --cached -- . ${TEST_FILE_EXCLUDES}`,\n }\n },\n\n async preflight() {\n await preflightVenvImports({\n modules: ['swebench'],\n requireDocker: true,\n fix:\n `Fix: (1) python3 -m venv bench/.venv && bench/.venv/bin/pip install swebench ; ` +\n `(2) ensure the Docker daemon is running (the judge builds per-instance images).`,\n })\n },\n\n async loadTasks(opts: LoadOptions = {}) {\n const limit = opts.limit ?? 10\n const split = opts.split ?? 'test'\n // Dump instances as JSON via the datasets loader (HF download on first run).\n const script = `\nimport json, sys\nfrom datasets import load_dataset\nds = load_dataset(${JSON.stringify(DATASET)}, split=${JSON.stringify(split)})\nids = set(json.loads(sys.argv[1])) if len(sys.argv) > 1 and sys.argv[1] else None\nout = []\nfor r in ds:\n if ids is not None and r[\"instance_id\"] not in ids:\n continue\n out.append({\n \"instance_id\": r[\"instance_id\"], \"repo\": r[\"repo\"], \"base_commit\": r[\"base_commit\"],\n \"problem_statement\": r[\"problem_statement\"], \"patch\": r[\"patch\"], \"test_patch\": r[\"test_patch\"],\n \"FAIL_TO_PASS\": r[\"FAIL_TO_PASS\"], \"PASS_TO_PASS\": r[\"PASS_TO_PASS\"],\n \"version\": r.get(\"version\"), \"environment_setup_commit\": r.get(\"environment_setup_commit\"),\n })\n if ids is None and len(out) >= ${limit}:\n break\nprint(json.dumps(out))\n`\n const stdout = await runVenvPython(script, [opts.ids ? JSON.stringify(opts.ids) : ''])\n const rows = JSON.parse(stdout) as Array<Record<string, unknown>>\n return rows.map(\n (r): BenchTask => ({\n id: String(r.instance_id),\n split,\n prompt: [\n `Repository: ${r.repo} @ ${r.base_commit}`,\n '',\n `The repository is ALREADY cloned at ${SWE_REPO_DIR}, checked out at commit ${r.base_commit}. Work there directly (\\`cd ${SWE_REPO_DIR}\\`); do not re-clone.`,\n '',\n 'Resolve this issue by editing the repository SOURCE so the failing tests pass without breaking the passing ones. Do NOT edit test files — the evaluation runs hidden tests on a fresh checkout, so editing tests does not count. Keep the change minimal and confined to the cloned repo.',\n 'Work iteratively: reproduce the issue, implement the fix in the source, and re-run the relevant tests until they pass. You do NOT need to print the diff — the harness reads your committed edits directly from the repo.',\n '',\n '--- Issue ---',\n String(r.problem_statement ?? ''),\n ].join('\\n'),\n metadata: r,\n }),\n )\n },\n\n async goldArtifact(task: BenchTask) {\n const gold = task.metadata?.patch\n return typeof gold === 'string' ? gold : undefined\n },\n\n async judge(task: BenchTask, artifact: string): Promise<BenchScore> {\n const runId = safeRunId('bench', task.id)\n const capture = options.captureEvaluatorArtifacts?.({\n taskId: task.id,\n runId,\n attemptSequence: ++attemptSequence,\n })\n return runStagedJudge({\n tmpPrefix: 'swebench-',\n ...(options.timeoutMs === undefined ? {} : { timeoutMs: options.timeoutMs }),\n ...(capture === undefined ? {} : { capture }),\n // Debug: retain the staged dir (holds swebench's per-instance apply/run\n // logs) for post-mortem when SWEBENCH_KEEP_TMP is set. Off by default.\n ...(process.env.SWEBENCH_KEEP_TMP ? { keepTmp: true } : {}),\n async stage(dir) {\n await stageFile(\n join(dir, 'preds.json'),\n JSON.stringify([\n { instance_id: task.id, model_name_or_path: 'agent-runtime-bench', model_patch: artifact },\n ]),\n )\n },\n // The official evaluation harness. Pulls/builds the instance image, applies\n // the patch, runs the test spec, writes a per-run report JSON in cwd.\n argv: (dir) => sweEvaluationArgv({\n predictionsPath: join(dir, 'preds.json'),\n runId,\n instanceId: task.id,\n cacheLevel,\n namespace: scorerNamespace(),\n }),\n async parseReport(dir) {\n // Report file: agent-runtime-bench.<run_id>.json\n const report = await readJsonReport<SweReport>(join(dir, `agent-runtime-bench.${runId}.json`))\n return scoreSweReport(task.id, report)\n },\n })\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;AAiCA,MAAM,eAAe;;;;;;;;;;AAWrB,MAAa,iBAAwC,EACnD,MAAM,QAAQ;CACZ,IAAI,OAAO;CACX,KAAK,MAAM,MAAM,QAAQ;EACvB,MAAM,IAAK,IAA2C;EACtD,MAAM,IAAI,GAAG,aAAa,GAAG,QAAQ,GAAG;EACxC,IAAI,OAAO,MAAM,YAAY,EAAE,SAAS,GAAG,OAAO;CACpD;CAKA,QADa,CADG,GAAG,KAAK,SAAS,uCAAuC,CACtD,CAAC,CAAC,GAAG,EAAE,CAAC,GAAG,MACb,KAAA,CAAM,KAAK;AAC7B,EACF;AAEA,MAAM,UAAU;AAsBhB,MAAM,mCAAmB,IAAI,IAAwB;CAAC;CAAQ;CAAQ;CAAO;AAAU,CAAC;AAExF,SAAS,kBAAuC;CAC9C,MAAM,YAAY,QAAQ,IAAI,sBAAsB;CACpD,IAAI,cAAc,cAAc,cAAc,QAC5C,MAAM,IAAI,MAAM,kDAAkD,UAAU,EAAE;CAEhF,OAAO;AACT;AACA,MAAM,qBAAqB;CACzB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC,CAAC,KAAK,GAAG;AAaV,SAAS,UAAU,QAAiC,KAAgC;CAClF,MAAM,QAAQ,OAAO;CACrB,IAAI,UAAU,KAAA,GAAW,OAAO,CAAC;CACjC,IAAI,CAAC,MAAM,QAAQ,KAAK,KAAK,MAAM,MAAM,UAAU,OAAO,UAAU,QAAQ,GAC1E,MAAM,IAAI,MAAM,wBAAwB,KAAK;CAE/C,OAAO;AACT;;AAGA,SAAgB,eAAe,QAAgB,OAA4B;CACzE,IAAI,CAAC,SAAS,OAAO,UAAU,YAAY,MAAM,QAAQ,KAAK,GAC5D,MAAM,IAAI,MAAM,qCAAqC;CAEvD,MAAM,SAAS;CACf,MAAM,YAAY;EAChB,UAAU,UAAU,QAAQ,cAAc;EAC1C,YAAY,UAAU,QAAQ,gBAAgB;EAC9C,YAAY,UAAU,QAAQ,iBAAiB;EAC/C,WAAW,UAAU,QAAQ,eAAe;EAC5C,YAAY,UAAU,QAAQ,gBAAgB;EAC9C,OAAO,UAAU,QAAQ,WAAW;CACtC;CACA,MAAM,YAAY,UAAU,QAAQ,eAAe;CAEnD,IADkB,OAAO,OAAO,SAAS,CAAC,CAAC,KAEjC,CAAC,CAAC,MAAM,OAAO,OAAO,MAAM,KAChC,UAAU,SAAS,MAAM,UAAU,WAAW,KAAK,UAAU,OAAO,SAExE,MAAM,IAAI,MAAM,2CAA2C,QAAQ;CAErE,IAAI,UAAU,MAAM,SAAS,MAAM,KAAK,UAAU,WAAW,SAAS,MAAM,GAC1E,MAAM,IAAI,MAAM,mCAAmC,QAAQ;CAE7D,MAAM,WAAW;EACf,UAAU,SAAS,SAAS,MAAM;EAClC,UAAU,WAAW,SAAS,MAAM;EACpC,UAAU,WAAW,SAAS,MAAM;CACtC;CACA,IAAI,SAAS,OAAO,OAAO,CAAC,CAAC,WAAW,GACtC,MAAM,IAAI,MAAM,+CAA+C,QAAQ;CAEzE,KAAK,SAAS,MAAM,SAAS,OAAO,CAAC,UAAU,UAAU,SAAS,MAAM,GACtE,MAAM,IAAI,MAAM,sDAAsD,QAAQ;CAEhF,MAAM,WAAW,SAAS;CAC1B,OAAO;EAAE;EAAU,OAAO,WAAW,IAAI;EAAG,QAAQ,KAAK,UAAU,MAAM;CAAE;AAC7E;AAEA,SAAgB,kBAAkB,MAMrB;CACX,OAAO;EACL;EAAM;EACN;EAAkB;EAClB;EAAsB,KAAK;EAC3B;EAAY,KAAK;EACjB;EAAkB,KAAK;EACvB;EAAiB;EACjB;EAAe,KAAK,aAAa,gBAAgB;EACjD;EAAiB,KAAK;CACxB;AACF;AAEA,SAAS,WAAW,OAAuB;CACzC,OAAO,IAAI,MAAM,QAAQ,MAAM,OAAO,EAAE;AAC1C;AAEA,SAAS,YAAY,MAAiD;CACpE,MAAM,OAAO,OAAO,KAAK,UAAU,QAAQ,EAAE;CAC7C,MAAM,OAAO,OAAO,KAAK,UAAU,eAAe,EAAE;CACpD,IAAI,CAAC,qCAAqC,KAAK,IAAI,GACjD,MAAM,IAAI,MAAM,wCAAwC,KAAK,GAAG,IAAI,MAAM;CAE5E,IAAI,CAAC,oBAAoB,KAAK,IAAI,GAChC,MAAM,IAAI,MAAM,+CAA+C,KAAK,GAAG,IAAI,MAAM;CAEnF,OAAO;EAAE;EAAM;CAAK;AACtB;AAEA,SAAgB,sBAAsB,UAAkC,CAAC,GAAqB;CAC5F,IACE,QAAQ,cAAc,KAAA,MAClB,CAAC,OAAO,cAAc,QAAQ,SAAS,KAAK,QAAQ,aAAa,IACrE,MAAM,IAAI,MAAM,iDAAiD;CACnE,MAAM,aAAa,QAAQ,cAAc;CACzC,IAAI,CAAC,iBAAiB,IAAI,UAAU,GAAG,MAAM,IAAI,MAAM,+BAA+B;CACtF,IACE,QAAQ,8BAA8B,KAAA,KACnC,OAAO,QAAQ,8BAA8B,YAChD,MAAM,IAAI,MAAM,yDAAyD;CAC3E,IAAI,kBAAkB;CACtB,OAAO;EACL,MAAM;EACN,QAAQ;EAWR,SAAS,MAAM;GACb,MAAM,EAAE,MAAM,SAAS,YAAY,IAAI;GACvC,OAAO,EACL,SAAS,UAAU,WAAW,YAAY,EAAE,wBAAwB,WAAW,sBAAsB,MAAM,EAAE,GAAG,WAAW,YAAY,EAAE,aAAa,WAAW,YAAY,EAAE,oBAAoB,WAAW,IAAI,IACpN;EACF;EACA,aAAa;GACX,OAAO,EACL,SAAS,UAAU,WAAW,YAAY,EAAE,oBAAoB,WAAW,YAAY,EAAE,sBAAsB,qBACjH;EACF;EAEA,MAAM,YAAY;GAChB,MAAM,qBAAqB;IACzB,SAAS,CAAC,UAAU;IACpB,eAAe;IACf,KACE;GAEJ,CAAC;EACH;EAEA,MAAM,UAAU,OAAoB,CAAC,GAAG;GACtC,MAAM,QAAQ,KAAK,SAAS;GAC5B,MAAM,QAAQ,KAAK,SAAS;GAqB5B,MAAM,SAAS,MAAM,cAAc;;;oBAhBrB,KAAK,UAAU,OAAO,EAAE,UAAU,KAAK,UAAU,KAAK,EAAE;;;;;;;;;;;;qCAYvC,MAAM;;;GAIM,CAAC,KAAK,MAAM,KAAK,UAAU,KAAK,GAAG,IAAI,EAAE,CAAC;GAErF,OADa,KAAK,MAAM,MACd,CAAC,CAAC,KACT,OAAkB;IACjB,IAAI,OAAO,EAAE,WAAW;IACxB;IACA,QAAQ;KACN,eAAe,EAAE,KAAK,KAAK,EAAE;KAC7B;KACA,uCAAuC,aAAa,0BAA0B,EAAE,YAAY,8BAA8B,aAAa;KACvI;KACA;KACA;KACA;KACA;KACA,OAAO,EAAE,qBAAqB,EAAE;IAClC,CAAC,CAAC,KAAK,IAAI;IACX,UAAU;GACZ,EACF;EACF;EAEA,MAAM,aAAa,MAAiB;GAClC,MAAM,OAAO,KAAK,UAAU;GAC5B,OAAO,OAAO,SAAS,WAAW,OAAO,KAAA;EAC3C;EAEA,MAAM,MAAM,MAAiB,UAAuC;GAClE,MAAM,QAAQ,UAAU,SAAS,KAAK,EAAE;GACxC,MAAM,UAAU,QAAQ,4BAA4B;IAClD,QAAQ,KAAK;IACb;IACA,iBAAiB,EAAE;GACrB,CAAC;GACD,OAAO,eAAe;IACpB,WAAW;IACX,GAAI,QAAQ,cAAc,KAAA,IAAY,CAAC,IAAI,EAAE,WAAW,QAAQ,UAAU;IAC1E,GAAI,YAAY,KAAA,IAAY,CAAC,IAAI,EAAE,QAAQ;IAG3C,GAAI,QAAQ,IAAI,oBAAoB,EAAE,SAAS,KAAK,IAAI,CAAC;IACzD,MAAM,MAAM,KAAK;KACf,MAAM,UACJ,KAAK,KAAK,YAAY,GACtB,KAAK,UAAU,CACb;MAAE,aAAa,KAAK;MAAI,oBAAoB;MAAuB,aAAa;KAAS,CAC3F,CAAC,CACH;IACF;IAGA,OAAO,QAAQ,kBAAkB;KAC/B,iBAAiB,KAAK,KAAK,YAAY;KACvC;KACA,YAAY,KAAK;KACjB;KACA,WAAW,gBAAgB;IAC7B,CAAC;IACD,MAAM,YAAY,KAAK;KAErB,MAAM,SAAS,MAAM,eAA0B,KAAK,KAAK,uBAAuB,MAAM,MAAM,CAAC;KAC7F,OAAO,eAAe,KAAK,IAAI,MAAM;IACvC;GACF,CAAC;EACH;CACF;AACF"}
|
|
1
|
+
{"version":3,"file":"swe-bench.js","names":[],"sources":["../../src/benchmarks/swe-bench.ts"],"sourcesContent":["/**\n * SWE-bench Verified adapter. Worker artifact = a unified-diff patch. Judge =\n * the official `swebench` harness: apply the patch in the instance's Docker\n * image, run FAIL_TO_PASS + PASS_TO_PASS, report `resolved`. Fully deterministic\n * — no LLM judge.\n *\n * Requires: the bench `.venv` with `swebench` installed + a running Docker\n * daemon (per-instance images are pulled/built on first run).\n *\n * Process/Docker/report plumbing is shared via ./_harness; this file owns the\n * SWE-specific pieces: the patch OutputAdapter, the dataset dump, and the\n * predictions-file → run_evaluation argv → report-shape mapping.\n */\n\nimport { join } from 'node:path'\nimport type { OutputAdapter } from '@tangle-network/agent-runtime/kernel'\nimport {\n preflightVenvImports,\n readJsonReport,\n runStagedJudge,\n runVenvPython,\n safeRunId,\n stageFile,\n type StagedRunCaptureSpec,\n} from './_harness'\nimport type { BenchmarkAdapter, BenchScore, BenchTask, LoadOptions } from './types'\n\n/** Root-level directories are not writable in every sandbox; use the session workspace. */\nconst SWE_REPO_DIR = './swe-bench-repo'\n\n/**\n * The SWE deliverable's FALLBACK parser, from the agent's event STREAM.\n *\n * The PRIMARY deliverable is `boxExtract` below: a `git diff` of the agent's\n * actual edits, read from the cloned repo's STATE inside the box (standard\n * SWE-bench practice). This event-stream parse only runs when that diff is empty\n * — a model that edited the source correctly but never printed a fenced diff (the\n * exact failure this replaces) still scores off its real changes, not its prose.\n */\nexport const swePatchOutput: OutputAdapter<string> = {\n parse(events) {\n let text = ''\n for (const ev of events) {\n const d = (ev as { data?: Record<string, unknown> })?.data\n const t = d?.finalText ?? d?.text ?? d?.result\n if (typeof t === 'string' && t.length > 0) text = t\n }\n // Last ```diff/```patch fenced block (the contract the prompt asks for);\n // fall back to the raw text so a fence-less but valid diff still reaches the judge.\n const fences = [...text.matchAll(/```(?:diff|patch)?\\s*\\n([\\s\\S]*?)```/g)]\n const last = fences.at(-1)?.[1]\n return (last ?? text).trim()\n },\n}\n\nconst DATASET = 'princeton-nlp/SWE-bench_Verified'\nexport type SweBenchCacheLevel = 'none' | 'base' | 'env' | 'instance'\n\nexport interface SweBenchArtifactCaptureContext {\n readonly taskId: string\n readonly runId: string\n /** One-based sequence unique within this adapter instance. */\n readonly attemptSequence: number\n}\n\nexport interface SweBenchAdapterOptions {\n readonly timeoutMs?: number\n readonly cacheLevel?: SweBenchCacheLevel\n /**\n * Return a unique destination for any attempt whose complete official\n * evaluator directory and process logs should be retained.\n */\n readonly captureEvaluatorArtifacts?: (\n context: SweBenchArtifactCaptureContext,\n ) => StagedRunCaptureSpec | undefined\n}\n\nconst SWE_CACHE_LEVELS = new Set<SweBenchCacheLevel>(['none', 'base', 'env', 'instance'])\n\nfunction scorerNamespace(): 'swebench' | 'none' {\n const namespace = process.env.SWEBENCH_NAMESPACE ?? 'swebench'\n if (namespace !== 'swebench' && namespace !== 'none') {\n throw new Error(`SWEBENCH_NAMESPACE must be swebench|none, got \"${namespace}\"`)\n }\n return namespace\n}\nconst TEST_FILE_EXCLUDES = [\n \"':(exclude,glob)**/tests/**'\",\n \"':(exclude,glob)**/test/**'\",\n \"':(exclude,glob)test_*.py'\",\n \"':(exclude,glob)**/test_*.py'\",\n \"':(exclude,glob)*_test.py'\",\n \"':(exclude,glob)**/*_test.py'\",\n \"':(exclude,glob)conftest.py'\",\n \"':(exclude,glob)**/conftest.py'\",\n].join(' ')\n\ninterface SweReport {\n resolved_instances?: number\n resolved_ids?: string[]\n unresolved_ids?: string[]\n empty_patch_ids?: string[]\n completed_ids?: string[]\n incomplete_ids?: string[]\n error_ids?: string[]\n submitted_ids?: string[]\n}\n\nfunction stringIds(report: Record<string, unknown>, key: keyof SweReport): string[] {\n const value = report[key]\n if (value === undefined) return []\n if (!Array.isArray(value) || value.some((entry) => typeof entry !== 'string')) {\n throw new Error(`swe-bench: malformed ${key}`)\n }\n return value\n}\n\n/** Convert one official report into a score without turning evaluator failures into agent failures. */\nexport function scoreSweReport(taskId: string, value: unknown): BenchScore {\n if (!value || typeof value !== 'object' || Array.isArray(value)) {\n throw new Error('swe-bench: report must be an object')\n }\n const report = value as Record<string, unknown>\n const statusIds = {\n resolved: stringIds(report, 'resolved_ids'),\n unresolved: stringIds(report, 'unresolved_ids'),\n emptyPatch: stringIds(report, 'empty_patch_ids'),\n completed: stringIds(report, 'completed_ids'),\n incomplete: stringIds(report, 'incomplete_ids'),\n error: stringIds(report, 'error_ids'),\n }\n const submitted = stringIds(report, 'submitted_ids')\n const mentioned = Object.values(statusIds).flat()\n if (\n mentioned.some((id) => id !== taskId)\n || (submitted.length > 0 && (submitted.length !== 1 || submitted[0] !== taskId))\n ) {\n throw new Error(`swe-bench: report identity mismatch for ${taskId}`)\n }\n if (statusIds.error.includes(taskId) || statusIds.incomplete.includes(taskId)) {\n throw new Error(`swe-bench: evaluator failed for ${taskId}`)\n }\n const outcomes = [\n statusIds.resolved.includes(taskId),\n statusIds.unresolved.includes(taskId),\n statusIds.emptyPatch.includes(taskId),\n ]\n if (outcomes.filter(Boolean).length !== 1) {\n throw new Error(`swe-bench: report has no unique outcome for ${taskId}`)\n }\n if ((outcomes[0] || outcomes[1]) && !statusIds.completed.includes(taskId)) {\n throw new Error(`swe-bench: report lacks a completed evaluation for ${taskId}`)\n }\n const resolved = outcomes[0]\n return { resolved, score: resolved ? 1 : 0, detail: JSON.stringify(report) }\n}\n\nexport function sweEvaluationArgv(args: {\n readonly predictionsPath: string\n readonly runId: string\n readonly instanceId: string\n readonly cacheLevel: SweBenchCacheLevel\n readonly namespace?: 'swebench' | 'none'\n}): string[] {\n return sweEvaluationArgvForHarness(args, true)\n}\n\nfunction sweEvaluationArgvForHarness(args: {\n readonly predictionsPath: string\n readonly runId: string\n readonly instanceId: string\n readonly cacheLevel: SweBenchCacheLevel\n readonly namespace?: 'swebench' | 'none'\n}, legacyFlags: boolean): string[] {\n const common = [\n '-m', 'swebench.harness.run_evaluation',\n '--dataset_name', DATASET,\n '--predictions_path', args.predictionsPath,\n '--run_id', args.runId,\n '--instance_ids', args.instanceId,\n '--max_workers', '1',\n ]\n if (!legacyFlags) return common\n return [...common, '--namespace', args.namespace ?? scorerNamespace(), '--cache_level', args.cacheLevel]\n}\n\nfunction shellQuote(value: string): string {\n return `'${value.replace(/'/g, `'\\\\''`)}'`\n}\n\nfunction sweMetadata(task: BenchTask): { repo: string; base: string } {\n const repo = String(task.metadata?.repo ?? '')\n const base = String(task.metadata?.base_commit ?? '')\n if (!/^[A-Za-z0-9_.-]+\\/[A-Za-z0-9_.-]+$/.test(repo)) {\n throw new Error(`swe-bench: invalid repo metadata for ${task.id}: ${repo}`)\n }\n if (!/^[0-9a-f]{7,40}$/i.test(base)) {\n throw new Error(`swe-bench: invalid base_commit metadata for ${task.id}: ${base}`)\n }\n return { repo, base }\n}\n\nexport function createSweBenchAdapter(options: SweBenchAdapterOptions = {}): BenchmarkAdapter {\n if (\n options.timeoutMs !== undefined\n && (!Number.isSafeInteger(options.timeoutMs) || options.timeoutMs <= 0)\n ) throw new Error('swe-bench: timeoutMs must be a positive integer')\n const cacheLevel = options.cacheLevel ?? 'env'\n if (!SWE_CACHE_LEVELS.has(cacheLevel)) throw new Error('swe-bench: invalid cacheLevel')\n if (\n options.captureEvaluatorArtifacts !== undefined\n && typeof options.captureEvaluatorArtifacts !== 'function'\n ) throw new Error('swe-bench: captureEvaluatorArtifacts must be a function')\n let attemptSequence = 0\n let legacyFlags: boolean | undefined\n return {\n name: 'swe-bench-verified',\n output: swePatchOutput,\n\n // Extract the patch from repo STATE, not printed text: stage every edit the\n // agent made in the cloned repo and diff it against the checked-out\n // base_commit (`HEAD`). Test files are excluded — the judge applies the gold\n // `test_patch` itself, so an agent edit to a test would collide on apply. The\n // paths come out `a/<repo-relative>` (cwd = repo root), matching the gold\n // patch format the swebench judge's `git apply` expects.\n // Pre-stage: clone the instance repo at base_commit into SWE_REPO_DIR so the\n // agent only edits (the harness owns the checkout — a stochastic model can't be\n // trusted to clone to an exact path). `--quiet` keeps the exec output small.\n boxSetup(task) {\n const { repo, base } = sweMetadata(task)\n return {\n command: `rm -rf ${shellQuote(SWE_REPO_DIR)} && git clone --quiet ${shellQuote(`https://github.com/${repo}`)} ${shellQuote(SWE_REPO_DIR)} && git -C ${shellQuote(SWE_REPO_DIR)} checkout --quiet ${shellQuote(base)}`,\n }\n },\n boxExtract() {\n return {\n command: `git -C ${shellQuote(SWE_REPO_DIR)} add -A && git -C ${shellQuote(SWE_REPO_DIR)} diff --cached -- . ${TEST_FILE_EXCLUDES}`,\n }\n },\n\n async preflight() {\n await preflightVenvImports({\n modules: ['swebench'],\n requireDocker: true,\n fix:\n `Fix: (1) python3 -m venv bench/.venv && bench/.venv/bin/pip install 'swebench>=4,<6' ; ` +\n `(2) ensure the Docker daemon is running (the judge builds per-instance images).`,\n })\n legacyFlags = await detectLegacyFlags()\n },\n\n async loadTasks(opts: LoadOptions = {}) {\n const limit = opts.limit ?? 10\n const split = opts.split ?? 'test'\n // Dump instances as JSON via the datasets loader (HF download on first run).\n const script = `\nimport json, sys\nfrom datasets import load_dataset\nds = load_dataset(${JSON.stringify(DATASET)}, split=${JSON.stringify(split)})\nids = set(json.loads(sys.argv[1])) if len(sys.argv) > 1 and sys.argv[1] else None\nout = []\nfor r in ds:\n if ids is not None and r[\"instance_id\"] not in ids:\n continue\n out.append({\n \"instance_id\": r[\"instance_id\"], \"repo\": r[\"repo\"], \"base_commit\": r[\"base_commit\"],\n \"problem_statement\": r[\"problem_statement\"], \"patch\": r[\"patch\"], \"test_patch\": r[\"test_patch\"],\n \"FAIL_TO_PASS\": r[\"FAIL_TO_PASS\"], \"PASS_TO_PASS\": r[\"PASS_TO_PASS\"],\n \"version\": r.get(\"version\"), \"environment_setup_commit\": r.get(\"environment_setup_commit\"),\n })\n if ids is None and len(out) >= ${limit}:\n break\nprint(json.dumps(out))\n`\n const stdout = await runVenvPython(script, [opts.ids ? JSON.stringify(opts.ids) : ''])\n const rows = JSON.parse(stdout) as Array<Record<string, unknown>>\n return rows.map(\n (r): BenchTask => ({\n id: String(r.instance_id),\n split,\n prompt: [\n `Repository: ${r.repo} @ ${r.base_commit}`,\n '',\n `The repository is ALREADY cloned at ${SWE_REPO_DIR} relative to your initial session workspace, checked out at commit ${r.base_commit}. Work there directly (\\`cd ${SWE_REPO_DIR}\\`); do not re-clone.`,\n '',\n 'Resolve this issue by editing the repository SOURCE so the failing tests pass without breaking the passing ones. Do NOT edit test files — the evaluation runs hidden tests on a fresh checkout, so editing tests does not count. Keep the change minimal and confined to the cloned repo.',\n 'Work iteratively: reproduce the issue, implement the fix in the source, and re-run the relevant tests until they pass. You do NOT need to print the diff — the harness reads your source edits directly from the repo, including uncommitted changes.',\n '',\n '--- Issue ---',\n String(r.problem_statement ?? ''),\n ].join('\\n'),\n metadata: r,\n }),\n )\n },\n\n async goldArtifact(task: BenchTask) {\n const gold = task.metadata?.patch\n return typeof gold === 'string' ? gold : undefined\n },\n\n async judge(task: BenchTask, artifact: string): Promise<BenchScore> {\n const useLegacyFlags = await ensureLegacyFlags()\n const runId = safeRunId('bench', task.id)\n const capture = options.captureEvaluatorArtifacts?.({\n taskId: task.id,\n runId,\n attemptSequence: ++attemptSequence,\n })\n return runStagedJudge({\n tmpPrefix: 'swebench-',\n ...(options.timeoutMs === undefined ? {} : { timeoutMs: options.timeoutMs }),\n ...(capture === undefined ? {} : { capture }),\n // Debug: retain the staged dir (holds swebench's per-instance apply/run\n // logs) for post-mortem when SWEBENCH_KEEP_TMP is set. Off by default.\n ...(process.env.SWEBENCH_KEEP_TMP ? { keepTmp: true } : {}),\n async stage(dir) {\n await stageFile(\n join(dir, 'preds.json'),\n JSON.stringify([\n { instance_id: task.id, model_name_or_path: 'agent-runtime-bench', model_patch: artifact },\n ]),\n )\n },\n // The official evaluation harness. Pulls/builds the instance image, applies\n // the patch, runs the test spec, writes a per-run report JSON in cwd.\n argv: (dir) => sweEvaluationArgvForHarness({\n predictionsPath: join(dir, 'preds.json'),\n runId,\n instanceId: task.id,\n cacheLevel,\n namespace: scorerNamespace(),\n }, useLegacyFlags),\n async parseReport(dir) {\n // Report file: agent-runtime-bench.<run_id>.json\n const report = await readJsonReport<SweReport>(join(dir, `agent-runtime-bench.${runId}.json`))\n return scoreSweReport(task.id, report)\n },\n })\n },\n }\n\n async function detectLegacyFlags(): Promise<boolean> {\n const script = `\nimport subprocess, sys\nresult = subprocess.run([sys.executable, '-m', 'swebench.harness.run_evaluation', '--help'], capture_output=True, text=True)\nif result.returncode != 0:\n raise SystemExit(result.stderr or result.stdout or str(result.returncode))\nprint('legacy' if '--namespace' in result.stdout and '--cache_level' in result.stdout else 'modern')\n`\n return (await runVenvPython(script)).trim() === 'legacy'\n }\n\n async function ensureLegacyFlags(): Promise<boolean> {\n if (legacyFlags === undefined) legacyFlags = await detectLegacyFlags()\n return legacyFlags\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;AA4BA,MAAM,eAAe;;;;;;;;;;AAWrB,MAAa,iBAAwC,EACnD,MAAM,QAAQ;CACZ,IAAI,OAAO;CACX,KAAK,MAAM,MAAM,QAAQ;EACvB,MAAM,IAAK,IAA2C;EACtD,MAAM,IAAI,GAAG,aAAa,GAAG,QAAQ,GAAG;EACxC,IAAI,OAAO,MAAM,YAAY,EAAE,SAAS,GAAG,OAAO;CACpD;CAKA,QADa,CADG,GAAG,KAAK,SAAS,uCAAuC,CACtD,CAAC,CAAC,GAAG,EAAE,CAAC,GAAG,MACb,KAAA,CAAM,KAAK;AAC7B,EACF;AAEA,MAAM,UAAU;AAsBhB,MAAM,mCAAmB,IAAI,IAAwB;CAAC;CAAQ;CAAQ;CAAO;AAAU,CAAC;AAExF,SAAS,kBAAuC;CAC9C,MAAM,YAAY,QAAQ,IAAI,sBAAsB;CACpD,IAAI,cAAc,cAAc,cAAc,QAC5C,MAAM,IAAI,MAAM,kDAAkD,UAAU,EAAE;CAEhF,OAAO;AACT;AACA,MAAM,qBAAqB;CACzB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC,CAAC,KAAK,GAAG;AAaV,SAAS,UAAU,QAAiC,KAAgC;CAClF,MAAM,QAAQ,OAAO;CACrB,IAAI,UAAU,KAAA,GAAW,OAAO,CAAC;CACjC,IAAI,CAAC,MAAM,QAAQ,KAAK,KAAK,MAAM,MAAM,UAAU,OAAO,UAAU,QAAQ,GAC1E,MAAM,IAAI,MAAM,wBAAwB,KAAK;CAE/C,OAAO;AACT;;AAGA,SAAgB,eAAe,QAAgB,OAA4B;CACzE,IAAI,CAAC,SAAS,OAAO,UAAU,YAAY,MAAM,QAAQ,KAAK,GAC5D,MAAM,IAAI,MAAM,qCAAqC;CAEvD,MAAM,SAAS;CACf,MAAM,YAAY;EAChB,UAAU,UAAU,QAAQ,cAAc;EAC1C,YAAY,UAAU,QAAQ,gBAAgB;EAC9C,YAAY,UAAU,QAAQ,iBAAiB;EAC/C,WAAW,UAAU,QAAQ,eAAe;EAC5C,YAAY,UAAU,QAAQ,gBAAgB;EAC9C,OAAO,UAAU,QAAQ,WAAW;CACtC;CACA,MAAM,YAAY,UAAU,QAAQ,eAAe;CAEnD,IADkB,OAAO,OAAO,SAAS,CAAC,CAAC,KAEjC,CAAC,CAAC,MAAM,OAAO,OAAO,MAAM,KAChC,UAAU,SAAS,MAAM,UAAU,WAAW,KAAK,UAAU,OAAO,SAExE,MAAM,IAAI,MAAM,2CAA2C,QAAQ;CAErE,IAAI,UAAU,MAAM,SAAS,MAAM,KAAK,UAAU,WAAW,SAAS,MAAM,GAC1E,MAAM,IAAI,MAAM,mCAAmC,QAAQ;CAE7D,MAAM,WAAW;EACf,UAAU,SAAS,SAAS,MAAM;EAClC,UAAU,WAAW,SAAS,MAAM;EACpC,UAAU,WAAW,SAAS,MAAM;CACtC;CACA,IAAI,SAAS,OAAO,OAAO,CAAC,CAAC,WAAW,GACtC,MAAM,IAAI,MAAM,+CAA+C,QAAQ;CAEzE,KAAK,SAAS,MAAM,SAAS,OAAO,CAAC,UAAU,UAAU,SAAS,MAAM,GACtE,MAAM,IAAI,MAAM,sDAAsD,QAAQ;CAEhF,MAAM,WAAW,SAAS;CAC1B,OAAO;EAAE;EAAU,OAAO,WAAW,IAAI;EAAG,QAAQ,KAAK,UAAU,MAAM;CAAE;AAC7E;AAEA,SAAgB,kBAAkB,MAMrB;CACX,OAAO,4BAA4B,MAAM,IAAI;AAC/C;AAEA,SAAS,4BAA4B,MAMlC,aAAgC;CACjC,MAAM,SAAS;EACb;EAAM;EACN;EAAkB;EAClB;EAAsB,KAAK;EAC3B;EAAY,KAAK;EACjB;EAAkB,KAAK;EACvB;EAAiB;CACnB;CACA,IAAI,CAAC,aAAa,OAAO;CACzB,OAAO;EAAC,GAAG;EAAQ;EAAe,KAAK,aAAa,gBAAgB;EAAG;EAAiB,KAAK;CAAU;AACzG;AAEA,SAAS,WAAW,OAAuB;CACzC,OAAO,IAAI,MAAM,QAAQ,MAAM,OAAO,EAAE;AAC1C;AAEA,SAAS,YAAY,MAAiD;CACpE,MAAM,OAAO,OAAO,KAAK,UAAU,QAAQ,EAAE;CAC7C,MAAM,OAAO,OAAO,KAAK,UAAU,eAAe,EAAE;CACpD,IAAI,CAAC,qCAAqC,KAAK,IAAI,GACjD,MAAM,IAAI,MAAM,wCAAwC,KAAK,GAAG,IAAI,MAAM;CAE5E,IAAI,CAAC,oBAAoB,KAAK,IAAI,GAChC,MAAM,IAAI,MAAM,+CAA+C,KAAK,GAAG,IAAI,MAAM;CAEnF,OAAO;EAAE;EAAM;CAAK;AACtB;AAEA,SAAgB,sBAAsB,UAAkC,CAAC,GAAqB;CAC5F,IACE,QAAQ,cAAc,KAAA,MAClB,CAAC,OAAO,cAAc,QAAQ,SAAS,KAAK,QAAQ,aAAa,IACrE,MAAM,IAAI,MAAM,iDAAiD;CACnE,MAAM,aAAa,QAAQ,cAAc;CACzC,IAAI,CAAC,iBAAiB,IAAI,UAAU,GAAG,MAAM,IAAI,MAAM,+BAA+B;CACtF,IACE,QAAQ,8BAA8B,KAAA,KACnC,OAAO,QAAQ,8BAA8B,YAChD,MAAM,IAAI,MAAM,yDAAyD;CAC3E,IAAI,kBAAkB;CACtB,IAAI;CACJ,OAAO;EACL,MAAM;EACN,QAAQ;EAWR,SAAS,MAAM;GACb,MAAM,EAAE,MAAM,SAAS,YAAY,IAAI;GACvC,OAAO,EACL,SAAS,UAAU,WAAW,YAAY,EAAE,wBAAwB,WAAW,sBAAsB,MAAM,EAAE,GAAG,WAAW,YAAY,EAAE,aAAa,WAAW,YAAY,EAAE,oBAAoB,WAAW,IAAI,IACpN;EACF;EACA,aAAa;GACX,OAAO,EACL,SAAS,UAAU,WAAW,YAAY,EAAE,oBAAoB,WAAW,YAAY,EAAE,sBAAsB,qBACjH;EACF;EAEA,MAAM,YAAY;GAChB,MAAM,qBAAqB;IACzB,SAAS,CAAC,UAAU;IACpB,eAAe;IACf,KACE;GAEJ,CAAC;GACD,cAAc,MAAM,kBAAkB;EACxC;EAEA,MAAM,UAAU,OAAoB,CAAC,GAAG;GACtC,MAAM,QAAQ,KAAK,SAAS;GAC5B,MAAM,QAAQ,KAAK,SAAS;GAqB5B,MAAM,SAAS,MAAM,cAAc;;;oBAhBrB,KAAK,UAAU,OAAO,EAAE,UAAU,KAAK,UAAU,KAAK,EAAE;;;;;;;;;;;;qCAYvC,MAAM;;;GAIM,CAAC,KAAK,MAAM,KAAK,UAAU,KAAK,GAAG,IAAI,EAAE,CAAC;GAErF,OADa,KAAK,MAAM,MACd,CAAC,CAAC,KACT,OAAkB;IACjB,IAAI,OAAO,EAAE,WAAW;IACxB;IACA,QAAQ;KACN,eAAe,EAAE,KAAK,KAAK,EAAE;KAC7B;KACA,uCAAuC,aAAa,qEAAqE,EAAE,YAAY,8BAA8B,aAAa;KAClL;KACA;KACA;KACA;KACA;KACA,OAAO,EAAE,qBAAqB,EAAE;IAClC,CAAC,CAAC,KAAK,IAAI;IACX,UAAU;GACZ,EACF;EACF;EAEA,MAAM,aAAa,MAAiB;GAClC,MAAM,OAAO,KAAK,UAAU;GAC5B,OAAO,OAAO,SAAS,WAAW,OAAO,KAAA;EAC3C;EAEA,MAAM,MAAM,MAAiB,UAAuC;GAClE,MAAM,iBAAiB,MAAM,kBAAkB;GAC/C,MAAM,QAAQ,UAAU,SAAS,KAAK,EAAE;GACxC,MAAM,UAAU,QAAQ,4BAA4B;IAClD,QAAQ,KAAK;IACb;IACA,iBAAiB,EAAE;GACrB,CAAC;GACD,OAAO,eAAe;IACpB,WAAW;IACX,GAAI,QAAQ,cAAc,KAAA,IAAY,CAAC,IAAI,EAAE,WAAW,QAAQ,UAAU;IAC1E,GAAI,YAAY,KAAA,IAAY,CAAC,IAAI,EAAE,QAAQ;IAG3C,GAAI,QAAQ,IAAI,oBAAoB,EAAE,SAAS,KAAK,IAAI,CAAC;IACzD,MAAM,MAAM,KAAK;KACf,MAAM,UACJ,KAAK,KAAK,YAAY,GACtB,KAAK,UAAU,CACb;MAAE,aAAa,KAAK;MAAI,oBAAoB;MAAuB,aAAa;KAAS,CAC3F,CAAC,CACH;IACF;IAGA,OAAO,QAAQ,4BAA4B;KACzC,iBAAiB,KAAK,KAAK,YAAY;KACvC;KACA,YAAY,KAAK;KACjB;KACA,WAAW,gBAAgB;IAC7B,GAAG,cAAc;IACjB,MAAM,YAAY,KAAK;KAErB,MAAM,SAAS,MAAM,eAA0B,KAAK,KAAK,uBAAuB,MAAM,MAAM,CAAC;KAC7F,OAAO,eAAe,KAAK,IAAI,MAAM;IACvC;GACF,CAAC;EACH;CACF;CAEA,eAAe,oBAAsC;EAQnD,QAAQ,MAAM,cAAc;;;;;;CAAM,EAAA,CAAG,KAAK,MAAM;CAClD;CAEA,eAAe,oBAAsC;EACnD,IAAI,gBAAgB,KAAA,GAAW,cAAc,MAAM,kBAAkB;EACrE,OAAO;CACT;AACF"}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-bench",
|
|
3
|
-
"version": "0.11.
|
|
3
|
+
"version": "0.11.3",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Benchmark adapters and execution for agent-runtime across coding, tool-use, RAG, memory, browser, and terminal tasks.",
|
|
6
6
|
"repository": {
|
|
@@ -25,11 +25,11 @@
|
|
|
25
25
|
}
|
|
26
26
|
},
|
|
27
27
|
"dependencies": {
|
|
28
|
-
"@tangle-network/agent-eval": ">=0.
|
|
28
|
+
"@tangle-network/agent-eval": ">=0.180.0 <0.181.0",
|
|
29
29
|
"@tangle-network/agent-interface": "^2.6.0",
|
|
30
|
-
"@tangle-network/agent-knowledge": "^15.0.
|
|
30
|
+
"@tangle-network/agent-knowledge": "^15.0.3",
|
|
31
31
|
"@tangle-network/sandbox": ">=0.36.4 <0.39.0",
|
|
32
|
-
"@tangle-network/agent-runtime": "^0.
|
|
32
|
+
"@tangle-network/agent-runtime": "^0.209.0"
|
|
33
33
|
},
|
|
34
34
|
"devDependencies": {
|
|
35
35
|
"@arethetypeswrong/cli": "0.18.5",
|
|
@@ -1,5 +1,9 @@
|
|
|
1
1
|
import assert from 'node:assert/strict'
|
|
2
2
|
import test from 'node:test'
|
|
3
|
+
import { execFileSync } from 'node:child_process'
|
|
4
|
+
import { mkdtempSync, mkdirSync, writeFileSync, rmSync, existsSync } from 'node:fs'
|
|
5
|
+
import { tmpdir } from 'node:os'
|
|
6
|
+
import { join } from 'node:path'
|
|
3
7
|
import { createSweBenchAdapter, scoreSweReport, sweEvaluationArgv } from './swe-bench'
|
|
4
8
|
|
|
5
9
|
const taskId = 'django__django-12345'
|
|
@@ -59,3 +63,48 @@ test('SWE evaluation command preserves the requested instance image', () => {
|
|
|
59
63
|
/invalid cacheLevel/,
|
|
60
64
|
)
|
|
61
65
|
})
|
|
66
|
+
|
|
67
|
+
test('SWE setup and extraction stay in the session workspace and exclude test edits', () => {
|
|
68
|
+
const root = mkdtempSync(join(tmpdir(), 'swe-workspace-'))
|
|
69
|
+
try {
|
|
70
|
+
const origin = join(root, 'origin')
|
|
71
|
+
const workspace = join(root, 'session')
|
|
72
|
+
mkdirSync(origin)
|
|
73
|
+
mkdirSync(workspace)
|
|
74
|
+
const git = (args: string[], input?: string) => execFileSync('git', args, { cwd: origin, input, encoding: 'utf8' }).trim()
|
|
75
|
+
git(['init', '--quiet'])
|
|
76
|
+
const blob = git(['hash-object', '-w', '--stdin'], 'before\n')
|
|
77
|
+
const tree = git(['mktree'], `100644 blob ${blob}\tsource.py\n`)
|
|
78
|
+
// Construct fixture history without changing the developer's Git identity or configuration.
|
|
79
|
+
const base = git(['hash-object', '-t', 'commit', '-w', '--stdin'],
|
|
80
|
+
`tree ${tree}\nauthor Fixture <fixture@example.invalid> 1 +0000\ncommitter Fixture <fixture@example.invalid> 1 +0000\n\nfixture\n`)
|
|
81
|
+
git(['update-ref', 'HEAD', base])
|
|
82
|
+
const task = { id: taskId, prompt: 'fix', metadata: { repo: 'fixture/repo', base_commit: base } }
|
|
83
|
+
const adapter = createSweBenchAdapter()
|
|
84
|
+
const setup = adapter.boxSetup!(task)
|
|
85
|
+
const extract = adapter.boxExtract!(task)
|
|
86
|
+
assert.match(setup.command, /^rm -rf '\.\//)
|
|
87
|
+
assert.equal(setup.cwd, undefined)
|
|
88
|
+
assert.equal(extract.cwd, undefined)
|
|
89
|
+
const env = {
|
|
90
|
+
...process.env,
|
|
91
|
+
GIT_CONFIG_COUNT: '2',
|
|
92
|
+
GIT_CONFIG_KEY_0: `url.file://${origin}.insteadOf`,
|
|
93
|
+
GIT_CONFIG_VALUE_0: 'https://github.com/fixture/repo',
|
|
94
|
+
GIT_CONFIG_KEY_1: 'protocol.file.allow',
|
|
95
|
+
GIT_CONFIG_VALUE_1: 'always',
|
|
96
|
+
}
|
|
97
|
+
execFileSync('sh', ['-c', setup.command], { cwd: workspace, env })
|
|
98
|
+
const repo = join(workspace, 'swe-bench-repo')
|
|
99
|
+
assert.ok(existsSync(join(repo, '.git')))
|
|
100
|
+
writeFileSync(join(repo, 'source.py'), 'after\n')
|
|
101
|
+
mkdirSync(join(repo, 'tests'))
|
|
102
|
+
writeFileSync(join(repo, 'tests', 'test_fix.py'), 'hidden-test-edit\n')
|
|
103
|
+
const patch = execFileSync('sh', ['-c', extract.command], { cwd: workspace, env, encoding: 'utf8' })
|
|
104
|
+
assert.match(patch, /diff --git a\/source.py b\/source.py/)
|
|
105
|
+
assert.match(patch, /\+after/)
|
|
106
|
+
assert.doesNotMatch(patch, /hidden-test-edit|test_fix/)
|
|
107
|
+
} finally {
|
|
108
|
+
rmSync(root, { recursive: true, force: true })
|
|
109
|
+
}
|
|
110
|
+
})
|
|
@@ -25,13 +25,8 @@ import {
|
|
|
25
25
|
} from './_harness'
|
|
26
26
|
import type { BenchmarkAdapter, BenchScore, BenchTask, LoadOptions } from './types'
|
|
27
27
|
|
|
28
|
-
/**
|
|
29
|
-
|
|
30
|
-
* source of truth shared by the prompt template (which tells the agent to clone
|
|
31
|
-
* here) and `boxExtract` (which runs `git diff` here after the shot) — so the
|
|
32
|
-
* harness always knows exactly where the agent's edits live, for any instance.
|
|
33
|
-
*/
|
|
34
|
-
const SWE_REPO_DIR = '/work'
|
|
28
|
+
/** Root-level directories are not writable in every sandbox; use the session workspace. */
|
|
29
|
+
const SWE_REPO_DIR = './swe-bench-repo'
|
|
35
30
|
|
|
36
31
|
/**
|
|
37
32
|
* The SWE deliverable's FALLBACK parser, from the agent's event STREAM.
|
|
@@ -167,16 +162,26 @@ export function sweEvaluationArgv(args: {
|
|
|
167
162
|
readonly cacheLevel: SweBenchCacheLevel
|
|
168
163
|
readonly namespace?: 'swebench' | 'none'
|
|
169
164
|
}): string[] {
|
|
170
|
-
return
|
|
165
|
+
return sweEvaluationArgvForHarness(args, true)
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
function sweEvaluationArgvForHarness(args: {
|
|
169
|
+
readonly predictionsPath: string
|
|
170
|
+
readonly runId: string
|
|
171
|
+
readonly instanceId: string
|
|
172
|
+
readonly cacheLevel: SweBenchCacheLevel
|
|
173
|
+
readonly namespace?: 'swebench' | 'none'
|
|
174
|
+
}, legacyFlags: boolean): string[] {
|
|
175
|
+
const common = [
|
|
171
176
|
'-m', 'swebench.harness.run_evaluation',
|
|
172
177
|
'--dataset_name', DATASET,
|
|
173
178
|
'--predictions_path', args.predictionsPath,
|
|
174
179
|
'--run_id', args.runId,
|
|
175
180
|
'--instance_ids', args.instanceId,
|
|
176
181
|
'--max_workers', '1',
|
|
177
|
-
'--namespace', args.namespace ?? scorerNamespace(),
|
|
178
|
-
'--cache_level', args.cacheLevel,
|
|
179
182
|
]
|
|
183
|
+
if (!legacyFlags) return common
|
|
184
|
+
return [...common, '--namespace', args.namespace ?? scorerNamespace(), '--cache_level', args.cacheLevel]
|
|
180
185
|
}
|
|
181
186
|
|
|
182
187
|
function shellQuote(value: string): string {
|
|
@@ -207,6 +212,7 @@ export function createSweBenchAdapter(options: SweBenchAdapterOptions = {}): Ben
|
|
|
207
212
|
&& typeof options.captureEvaluatorArtifacts !== 'function'
|
|
208
213
|
) throw new Error('swe-bench: captureEvaluatorArtifacts must be a function')
|
|
209
214
|
let attemptSequence = 0
|
|
215
|
+
let legacyFlags: boolean | undefined
|
|
210
216
|
return {
|
|
211
217
|
name: 'swe-bench-verified',
|
|
212
218
|
output: swePatchOutput,
|
|
@@ -237,9 +243,10 @@ export function createSweBenchAdapter(options: SweBenchAdapterOptions = {}): Ben
|
|
|
237
243
|
modules: ['swebench'],
|
|
238
244
|
requireDocker: true,
|
|
239
245
|
fix:
|
|
240
|
-
`Fix: (1) python3 -m venv bench/.venv && bench/.venv/bin/pip install swebench ; ` +
|
|
246
|
+
`Fix: (1) python3 -m venv bench/.venv && bench/.venv/bin/pip install 'swebench>=4,<6' ; ` +
|
|
241
247
|
`(2) ensure the Docker daemon is running (the judge builds per-instance images).`,
|
|
242
248
|
})
|
|
249
|
+
legacyFlags = await detectLegacyFlags()
|
|
243
250
|
},
|
|
244
251
|
|
|
245
252
|
async loadTasks(opts: LoadOptions = {}) {
|
|
@@ -274,10 +281,10 @@ print(json.dumps(out))
|
|
|
274
281
|
prompt: [
|
|
275
282
|
`Repository: ${r.repo} @ ${r.base_commit}`,
|
|
276
283
|
'',
|
|
277
|
-
`The repository is ALREADY cloned at ${SWE_REPO_DIR}, checked out at commit ${r.base_commit}. Work there directly (\`cd ${SWE_REPO_DIR}\`); do not re-clone.`,
|
|
284
|
+
`The repository is ALREADY cloned at ${SWE_REPO_DIR} relative to your initial session workspace, checked out at commit ${r.base_commit}. Work there directly (\`cd ${SWE_REPO_DIR}\`); do not re-clone.`,
|
|
278
285
|
'',
|
|
279
286
|
'Resolve this issue by editing the repository SOURCE so the failing tests pass without breaking the passing ones. Do NOT edit test files — the evaluation runs hidden tests on a fresh checkout, so editing tests does not count. Keep the change minimal and confined to the cloned repo.',
|
|
280
|
-
'Work iteratively: reproduce the issue, implement the fix in the source, and re-run the relevant tests until they pass. You do NOT need to print the diff — the harness reads your
|
|
287
|
+
'Work iteratively: reproduce the issue, implement the fix in the source, and re-run the relevant tests until they pass. You do NOT need to print the diff — the harness reads your source edits directly from the repo, including uncommitted changes.',
|
|
281
288
|
'',
|
|
282
289
|
'--- Issue ---',
|
|
283
290
|
String(r.problem_statement ?? ''),
|
|
@@ -293,6 +300,7 @@ print(json.dumps(out))
|
|
|
293
300
|
},
|
|
294
301
|
|
|
295
302
|
async judge(task: BenchTask, artifact: string): Promise<BenchScore> {
|
|
303
|
+
const useLegacyFlags = await ensureLegacyFlags()
|
|
296
304
|
const runId = safeRunId('bench', task.id)
|
|
297
305
|
const capture = options.captureEvaluatorArtifacts?.({
|
|
298
306
|
taskId: task.id,
|
|
@@ -316,13 +324,13 @@ print(json.dumps(out))
|
|
|
316
324
|
},
|
|
317
325
|
// The official evaluation harness. Pulls/builds the instance image, applies
|
|
318
326
|
// the patch, runs the test spec, writes a per-run report JSON in cwd.
|
|
319
|
-
argv: (dir) =>
|
|
327
|
+
argv: (dir) => sweEvaluationArgvForHarness({
|
|
320
328
|
predictionsPath: join(dir, 'preds.json'),
|
|
321
329
|
runId,
|
|
322
330
|
instanceId: task.id,
|
|
323
331
|
cacheLevel,
|
|
324
332
|
namespace: scorerNamespace(),
|
|
325
|
-
}),
|
|
333
|
+
}, useLegacyFlags),
|
|
326
334
|
async parseReport(dir) {
|
|
327
335
|
// Report file: agent-runtime-bench.<run_id>.json
|
|
328
336
|
const report = await readJsonReport<SweReport>(join(dir, `agent-runtime-bench.${runId}.json`))
|
|
@@ -331,4 +339,20 @@ print(json.dumps(out))
|
|
|
331
339
|
})
|
|
332
340
|
},
|
|
333
341
|
}
|
|
342
|
+
|
|
343
|
+
async function detectLegacyFlags(): Promise<boolean> {
|
|
344
|
+
const script = `
|
|
345
|
+
import subprocess, sys
|
|
346
|
+
result = subprocess.run([sys.executable, '-m', 'swebench.harness.run_evaluation', '--help'], capture_output=True, text=True)
|
|
347
|
+
if result.returncode != 0:
|
|
348
|
+
raise SystemExit(result.stderr or result.stdout or str(result.returncode))
|
|
349
|
+
print('legacy' if '--namespace' in result.stdout and '--cache_level' in result.stdout else 'modern')
|
|
350
|
+
`
|
|
351
|
+
return (await runVenvPython(script)).trim() === 'legacy'
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
async function ensureLegacyFlags(): Promise<boolean> {
|
|
355
|
+
if (legacyFlags === undefined) legacyFlags = await detectLegacyFlags()
|
|
356
|
+
return legacyFlags
|
|
357
|
+
}
|
|
334
358
|
}
|