@tangle-network/agent-bench 0.13.11 → 0.13.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +10 -0
- package/HARNESS.md +2 -0
- package/package.json +4 -4
- package/src/swe-local-proof.mts +72 -3
- package/src/terminal-compare.ts +9 -13
- package/tb_agents/opencode_router_agent.py +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,15 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.13.13
|
|
4
|
+
|
|
5
|
+
Require Knowledge 17.1.6 and Eval `>=0.191.0 <0.194.0` through the shared dependency catalog.
|
|
6
|
+
Benchmark behavior is unchanged.
|
|
7
|
+
|
|
8
|
+
## 0.13.12
|
|
9
|
+
|
|
10
|
+
Accept Eval 0.185 and 0.186 and require Knowledge 17.1.2 through the shared dependency catalog.
|
|
11
|
+
Benchmark behavior is unchanged.
|
|
12
|
+
|
|
3
13
|
## 0.13.11
|
|
4
14
|
|
|
5
15
|
Accept Eval 0.184 and require Knowledge 17.1.1 through the shared dependency catalog.
|
package/HARNESS.md
CHANGED
|
@@ -28,6 +28,8 @@ Use these labels literally. Do not promote one level into another in prose.
|
|
|
28
28
|
|
|
29
29
|
A localization score, output-shape check, LLM quality judge, or toy deterministic reward can be useful for development. None is a substitute for the benchmark's outcome evaluator.
|
|
30
30
|
|
|
31
|
+
The [September 24 public benchmark proof](evidence/public-agent-benchmarks-20260924/README.md) retains three officially graded agent tasks, traces, and validity limits across Terminal-Bench and SWE-bench Verified.
|
|
32
|
+
|
|
31
33
|
## Supported commands
|
|
32
34
|
|
|
33
35
|
### Package and integration contracts
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-bench",
|
|
3
|
-
"version": "0.13.
|
|
3
|
+
"version": "0.13.13",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Benchmark adapters and execution for agent-runtime across coding, tool-use, RAG, memory, browser, and terminal tasks.",
|
|
6
6
|
"repository": {
|
|
@@ -25,11 +25,11 @@
|
|
|
25
25
|
}
|
|
26
26
|
},
|
|
27
27
|
"dependencies": {
|
|
28
|
-
"@tangle-network/agent-eval": ">=0.
|
|
28
|
+
"@tangle-network/agent-eval": ">=0.191.0 <0.194.0",
|
|
29
29
|
"@tangle-network/agent-interface": "^2.11.0",
|
|
30
|
-
"@tangle-network/agent-knowledge": "^17.1.
|
|
30
|
+
"@tangle-network/agent-knowledge": "^17.1.6",
|
|
31
31
|
"@tangle-network/sandbox": ">=0.36.4 <0.48.0",
|
|
32
|
-
"@tangle-network/agent-runtime": "^0.
|
|
32
|
+
"@tangle-network/agent-runtime": "^0.277.0"
|
|
33
33
|
},
|
|
34
34
|
"devDependencies": {
|
|
35
35
|
"@arethetypeswrong/cli": "0.18.5",
|
package/src/swe-local-proof.mts
CHANGED
|
@@ -16,7 +16,7 @@
|
|
|
16
16
|
* node_modules/.bin/tsx bench/src/swe-local-proof.mts
|
|
17
17
|
*/
|
|
18
18
|
import { execFile } from 'node:child_process'
|
|
19
|
-
import { mkdtempSync, rmSync, writeFileSync } from 'node:fs'
|
|
19
|
+
import { appendFileSync, mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'
|
|
20
20
|
import { tmpdir } from 'node:os'
|
|
21
21
|
import { join } from 'node:path'
|
|
22
22
|
import { promisify } from 'node:util'
|
|
@@ -40,12 +40,24 @@ async function main(): Promise<void> {
|
|
|
40
40
|
// WITH-TOOLS arm: RUN_TOOL=1 exposes the jailed `run` tool + the run-aware prompt. Default OFF ⇒
|
|
41
41
|
// the read/edit-only baseline (glm-5.2 7/12) unchanged.
|
|
42
42
|
const enableRun = ['1', 'true', 'yes'].includes((process.env.RUN_TOOL ?? '').toLowerCase())
|
|
43
|
+
const artifactDir = process.env.RUN_ARTIFACT_DIR
|
|
44
|
+
if (artifactDir) mkdirSync(artifactDir, { recursive: true })
|
|
43
45
|
|
|
44
46
|
console.log(`═══ SWE-bench Verified — SEE-able LOCAL proof (no tangle sandbox) ═══`)
|
|
45
47
|
console.log(`model=${model} ids=${ids.join(',')} innerTurns=${innerTurns} maxTokens=${maxTokens} budget=${budget} runTool=${enableRun}`)
|
|
46
48
|
console.log(`router=${routerBaseUrl}`)
|
|
47
49
|
|
|
48
|
-
const { environment, tasks, adapter } = await createSweBenchEnvironment(ids.length, {
|
|
50
|
+
const { environment, tasks, adapter } = await createSweBenchEnvironment(ids.length, {
|
|
51
|
+
ids,
|
|
52
|
+
enableRun,
|
|
53
|
+
...(artifactDir ? {
|
|
54
|
+
adapterOptions: {
|
|
55
|
+
captureEvaluatorArtifacts: ({ taskId, attemptSequence }) => ({
|
|
56
|
+
destination: join(artifactDir, taskId, `judge-${attemptSequence}`),
|
|
57
|
+
}),
|
|
58
|
+
},
|
|
59
|
+
} : {}),
|
|
60
|
+
})
|
|
49
61
|
const workerProfile = withBenchProfile(
|
|
50
62
|
{
|
|
51
63
|
name: 'swe-local-proof-worker',
|
|
@@ -85,10 +97,27 @@ async function main(): Promise<void> {
|
|
|
85
97
|
const captured = new Map<string, Rec>()
|
|
86
98
|
const judged = new Map<string, BenchScore>()
|
|
87
99
|
const toolStats = new Map<string, { list: number; read: number; edit_ok: number; edit_fail: number; run: number; run_err: number }>()
|
|
100
|
+
let activeTaskId = ''
|
|
101
|
+
const recordTool = (name: string, args: unknown, result: unknown, error?: string) => {
|
|
102
|
+
if (!artifactDir || !activeTaskId) return
|
|
103
|
+
const taskDir = join(artifactDir, activeTaskId)
|
|
104
|
+
mkdirSync(taskDir, { recursive: true })
|
|
105
|
+
appendFileSync(join(taskDir, 'tools.jsonl'), JSON.stringify({
|
|
106
|
+
at: new Date().toISOString(), name, args,
|
|
107
|
+
...(error ? { error } : { result }),
|
|
108
|
+
}) + '\n')
|
|
109
|
+
}
|
|
88
110
|
const proxy: AgenticSurface = {
|
|
89
111
|
...environment,
|
|
90
112
|
async call(handle, name, args) {
|
|
91
|
-
|
|
113
|
+
let res: string
|
|
114
|
+
try {
|
|
115
|
+
res = await environment.call(handle, name, args)
|
|
116
|
+
} catch (error) {
|
|
117
|
+
recordTool(name, args, undefined, error instanceof Error ? error.message : String(error))
|
|
118
|
+
throw error
|
|
119
|
+
}
|
|
120
|
+
recordTool(name, args, res)
|
|
92
121
|
// Count tool usage per workspace so we can SEE whether the agent ever edited, and whether
|
|
93
122
|
// edits succeeded or bounced off old_string matching. handle.id keys the workspace.
|
|
94
123
|
const st = toolStats.get(handle.id) ?? { list: 0, read: 0, edit_ok: 0, edit_fail: 0, run: 0, run_err: 0 }
|
|
@@ -161,6 +190,7 @@ async function main(): Promise<void> {
|
|
|
161
190
|
|
|
162
191
|
let anyResolved = 0
|
|
163
192
|
for (const task of taskList) {
|
|
193
|
+
activeTaskId = task.id
|
|
164
194
|
const t0 = Date.now()
|
|
165
195
|
const r = await runAgentic({
|
|
166
196
|
surface: proxy,
|
|
@@ -171,6 +201,13 @@ async function main(): Promise<void> {
|
|
|
171
201
|
workerProfile,
|
|
172
202
|
analystProfile,
|
|
173
203
|
budget,
|
|
204
|
+
...(artifactDir ? { hooks: { onEvent(event: unknown) {
|
|
205
|
+
const taskDir = join(artifactDir, activeTaskId)
|
|
206
|
+
mkdirSync(taskDir, { recursive: true })
|
|
207
|
+
appendFileSync(join(taskDir, 'runtime-events.jsonl'), JSON.stringify(event, (_key, value) =>
|
|
208
|
+
value instanceof Error ? { name: value.name, message: value.message, stack: value.stack } : value,
|
|
209
|
+
) + '\n')
|
|
210
|
+
} } } : {}),
|
|
174
211
|
})
|
|
175
212
|
const rec = captured.get(task.id)
|
|
176
213
|
const st = toolStats.get(task.id)
|
|
@@ -186,6 +223,38 @@ async function main(): Promise<void> {
|
|
|
186
223
|
console.log(` patch APPLIED (git apply --check on clean base): ${rec?.applied ? 'YES' : 'NO'}${rec?.applyErr ? ` (${rec.applyErr})` : ''}`)
|
|
187
224
|
console.log(` swebench judge RESOLVED: ${resolved ? '1' : '0'}`)
|
|
188
225
|
if (rec?.score?.detail) console.log(` judge report: ${rec.score.detail.slice(0, 400)}`)
|
|
226
|
+
if (artifactDir) {
|
|
227
|
+
const taskDir = join(artifactDir, task.id)
|
|
228
|
+
mkdirSync(taskDir, { recursive: true })
|
|
229
|
+
writeFileSync(join(taskDir, 'patch.diff'), rec?.patch ?? '')
|
|
230
|
+
writeFileSync(join(taskDir, 'summary.json'), JSON.stringify({
|
|
231
|
+
benchmark: 'SWE-bench Verified',
|
|
232
|
+
taskId: task.id,
|
|
233
|
+
modelRequested: model,
|
|
234
|
+
prompt: task.userPrompt,
|
|
235
|
+
turnsLimit: innerTurns,
|
|
236
|
+
maxTokens,
|
|
237
|
+
budget,
|
|
238
|
+
enableRun,
|
|
239
|
+
shots: r.shots,
|
|
240
|
+
completions: r.completions,
|
|
241
|
+
tokens: r.tokens,
|
|
242
|
+
tokensKnown: r.tokensKnown,
|
|
243
|
+
billedCostUsd: null,
|
|
244
|
+
reportedUsd: r.usd,
|
|
245
|
+
reportedUsdKnown: r.usdKnown,
|
|
246
|
+
progression: r.progression,
|
|
247
|
+
runScore: r.score,
|
|
248
|
+
runResolved: r.resolved,
|
|
249
|
+
wallMs: Date.now() - t0,
|
|
250
|
+
toolStats: st ?? null,
|
|
251
|
+
patchBytes,
|
|
252
|
+
patchLines,
|
|
253
|
+
files,
|
|
254
|
+
patchApplies: rec?.applied ?? false,
|
|
255
|
+
score: rec?.score ?? null,
|
|
256
|
+
}, null, 2) + '\n')
|
|
257
|
+
}
|
|
189
258
|
}
|
|
190
259
|
console.log(`\n>>> resolved ${anyResolved}/${taskList.length}`)
|
|
191
260
|
}
|
package/src/terminal-compare.ts
CHANGED
|
@@ -58,7 +58,7 @@ const PROFILE = agentProfileSchema.parse({
|
|
|
58
58
|
systemPrompt:
|
|
59
59
|
'Solve the Terminal-Bench task completely in the provided container and verify the result before finishing.',
|
|
60
60
|
},
|
|
61
|
-
|
|
61
|
+
permissions: {
|
|
62
62
|
edit: 'allow',
|
|
63
63
|
bash: 'allow',
|
|
64
64
|
webfetch: 'allow',
|
|
@@ -360,14 +360,9 @@ async function captureRunRecord(
|
|
|
360
360
|
await appendRunRecord(CORPUS, record)
|
|
361
361
|
}
|
|
362
362
|
|
|
363
|
-
/**
|
|
364
|
-
*
|
|
365
|
-
*
|
|
366
|
-
* contends on the shared docker build backend and can return nonzero. Warming the
|
|
367
|
-
* cache serially makes the parallel rounds hit a warm cache and never race.
|
|
368
|
-
* Dataset-agnostic: derives the compose path from the tb cache layout
|
|
369
|
-
* (<cache>/<name>/<version>/<task>/docker-compose.yaml).
|
|
370
|
-
*/
|
|
363
|
+
/** Warm each task image serially before the concurrent fan-out. Terminal-Bench
|
|
364
|
+
* supplies compose variables only inside its own runner, so build the task's
|
|
365
|
+
* Dockerfile directly and let its later compose build reuse the cached layers. */
|
|
371
366
|
async function prebuildImages(taskIds: string[]): Promise<void> {
|
|
372
367
|
const [name, version] = DATASET.split('==')
|
|
373
368
|
if (!name || !version) {
|
|
@@ -376,18 +371,19 @@ async function prebuildImages(taskIds: string[]): Promise<void> {
|
|
|
376
371
|
}
|
|
377
372
|
const cacheRoot = join(homedir(), '.cache', 'terminal-bench', name, version)
|
|
378
373
|
for (const taskId of taskIds) {
|
|
379
|
-
const
|
|
374
|
+
const taskDir = join(cacheRoot, taskId)
|
|
375
|
+
const dockerfile = join(taskDir, 'Dockerfile')
|
|
380
376
|
try {
|
|
381
|
-
await stat(
|
|
377
|
+
await stat(dockerfile)
|
|
382
378
|
} catch {
|
|
383
|
-
console.log(` prebuild: no
|
|
379
|
+
console.log(` prebuild: no Dockerfile at ${dockerfile}; tb will materialize ${taskId} on first run`)
|
|
384
380
|
continue
|
|
385
381
|
}
|
|
386
382
|
process.stdout.write(` prebuild: ${taskId} … `)
|
|
387
383
|
await new Promise<void>((resolve, reject) => {
|
|
388
384
|
execFile(
|
|
389
385
|
'docker',
|
|
390
|
-
['
|
|
386
|
+
['build', '--tag', `tbprebuild-${taskId}`.toLowerCase().replace(/[^a-z0-9_.-]/g, '-'), '--file', dockerfile, taskDir],
|
|
391
387
|
{ maxBuffer: 1024 * 1024 * 256, timeout: ROUND_CAP_MS },
|
|
392
388
|
(err) => (err ? reject(new Error(`prebuild ${taskId} failed: ${err.message}`)) : resolve()),
|
|
393
389
|
)
|
|
@@ -202,7 +202,7 @@ class OpenCodeRouterAgent(OpenCodeAgent):
|
|
|
202
202
|
"$schema": "https://opencode.ai/config.json",
|
|
203
203
|
# Headless benchmark runs cannot answer interactive permission prompts.
|
|
204
204
|
# Keep this identical for raw and supervisor arms.
|
|
205
|
-
"permission": self._profile.get("
|
|
205
|
+
"permission": self._profile.get("permissions", {}),
|
|
206
206
|
"provider": {
|
|
207
207
|
self._provider: {
|
|
208
208
|
"npm": "@ai-sdk/openai-compatible",
|