@tangle-network/agent-bench 0.13.7 → 0.13.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,35 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.13.13
4
+
5
+ Require Knowledge 17.1.6 and Eval `>=0.191.0 <0.194.0` through the shared dependency catalog.
6
+ Benchmark behavior is unchanged.
7
+
8
+ ## 0.13.12
9
+
10
+ Accept Eval 0.185 and 0.186 and require Knowledge 17.1.2 through the shared dependency catalog.
11
+ Benchmark behavior is unchanged.
12
+
13
+ ## 0.13.11
14
+
15
+ Accept Eval 0.184 and require Knowledge 17.1.1 through the shared dependency catalog.
16
+ Benchmark behavior is unchanged.
17
+
18
+ ## 0.13.10
19
+
20
+ Accept Sandbox SDK 0.47 through the shared dependency catalog.
21
+ Benchmark behavior is unchanged.
22
+
23
+ ## 0.13.9
24
+
25
+ Accept agent-interface 2.11.0 through the shared dependency catalog.
26
+ Benchmark behavior is unchanged.
27
+
28
+ ## 0.13.8
29
+
30
+ Accept Sandbox SDK 0.46 through the shared dependency catalog.
31
+ Benchmark behavior is unchanged.
32
+
3
33
  ## 0.13.7
4
34
 
5
35
  Support Eval 0.183 and require Knowledge 17.1 through the shared dependency catalog.
package/HARNESS.md CHANGED
@@ -28,6 +28,8 @@ Use these labels literally. Do not promote one level into another in prose.
28
28
 
29
29
  A localization score, output-shape check, LLM quality judge, or toy deterministic reward can be useful for development. None is a substitute for the benchmark's outcome evaluator.
30
30
 
31
+ The [September 24 public benchmark proof](evidence/public-agent-benchmarks-20260924/README.md) retains three officially graded agent tasks, traces, and validity limits across Terminal-Bench and SWE-bench Verified.
32
+
31
33
  ## Supported commands
32
34
 
33
35
  ### Package and integration contracts
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-bench",
3
- "version": "0.13.7",
3
+ "version": "0.13.13",
4
4
  "type": "module",
5
5
  "description": "Benchmark adapters and execution for agent-runtime across coding, tool-use, RAG, memory, browser, and terminal tasks.",
6
6
  "repository": {
@@ -25,11 +25,11 @@
25
25
  }
26
26
  },
27
27
  "dependencies": {
28
- "@tangle-network/agent-eval": ">=0.183.0 <0.184.0",
29
- "@tangle-network/agent-interface": "^2.10.0",
30
- "@tangle-network/agent-knowledge": "^17.1.0",
31
- "@tangle-network/sandbox": ">=0.36.4 <0.46.0",
32
- "@tangle-network/agent-runtime": "^0.249.1"
28
+ "@tangle-network/agent-eval": ">=0.191.0 <0.194.0",
29
+ "@tangle-network/agent-interface": "^2.11.0",
30
+ "@tangle-network/agent-knowledge": "^17.1.6",
31
+ "@tangle-network/sandbox": ">=0.36.4 <0.48.0",
32
+ "@tangle-network/agent-runtime": "^0.277.0"
33
33
  },
34
34
  "devDependencies": {
35
35
  "@arethetypeswrong/cli": "0.18.5",
@@ -47,6 +47,11 @@ const makeRelativeSymlinkRepo = (): { root: string; source: string; destination:
47
47
  `core.hooksPath=${hooksDir}`,
48
48
  '-c',
49
49
  'commit.gpgsign=false',
50
+ // A commit starts `git maintenance run --auto` in the background, which creates and removes
51
+ // .git/objects/maintenance.lock while the test copies the checkout. The copy then fails with
52
+ // ENOENT on a file that vanished: 5 of 240 copies on Node 22.23 and 24.11, 0 of 400 without it.
53
+ '-c',
54
+ 'maintenance.auto=false',
50
55
  'commit',
51
56
  '--quiet',
52
57
  '-m',
@@ -16,7 +16,7 @@
16
16
  * node_modules/.bin/tsx bench/src/swe-local-proof.mts
17
17
  */
18
18
  import { execFile } from 'node:child_process'
19
- import { mkdtempSync, rmSync, writeFileSync } from 'node:fs'
19
+ import { appendFileSync, mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'
20
20
  import { tmpdir } from 'node:os'
21
21
  import { join } from 'node:path'
22
22
  import { promisify } from 'node:util'
@@ -40,12 +40,24 @@ async function main(): Promise<void> {
40
40
  // WITH-TOOLS arm: RUN_TOOL=1 exposes the jailed `run` tool + the run-aware prompt. Default OFF ⇒
41
41
  // the read/edit-only baseline (glm-5.2 7/12) unchanged.
42
42
  const enableRun = ['1', 'true', 'yes'].includes((process.env.RUN_TOOL ?? '').toLowerCase())
43
+ const artifactDir = process.env.RUN_ARTIFACT_DIR
44
+ if (artifactDir) mkdirSync(artifactDir, { recursive: true })
43
45
 
44
46
  console.log(`═══ SWE-bench Verified — SEE-able LOCAL proof (no tangle sandbox) ═══`)
45
47
  console.log(`model=${model} ids=${ids.join(',')} innerTurns=${innerTurns} maxTokens=${maxTokens} budget=${budget} runTool=${enableRun}`)
46
48
  console.log(`router=${routerBaseUrl}`)
47
49
 
48
- const { environment, tasks, adapter } = await createSweBenchEnvironment(ids.length, { ids, enableRun })
50
+ const { environment, tasks, adapter } = await createSweBenchEnvironment(ids.length, {
51
+ ids,
52
+ enableRun,
53
+ ...(artifactDir ? {
54
+ adapterOptions: {
55
+ captureEvaluatorArtifacts: ({ taskId, attemptSequence }) => ({
56
+ destination: join(artifactDir, taskId, `judge-${attemptSequence}`),
57
+ }),
58
+ },
59
+ } : {}),
60
+ })
49
61
  const workerProfile = withBenchProfile(
50
62
  {
51
63
  name: 'swe-local-proof-worker',
@@ -85,10 +97,27 @@ async function main(): Promise<void> {
85
97
  const captured = new Map<string, Rec>()
86
98
  const judged = new Map<string, BenchScore>()
87
99
  const toolStats = new Map<string, { list: number; read: number; edit_ok: number; edit_fail: number; run: number; run_err: number }>()
100
+ let activeTaskId = ''
101
+ const recordTool = (name: string, args: unknown, result: unknown, error?: string) => {
102
+ if (!artifactDir || !activeTaskId) return
103
+ const taskDir = join(artifactDir, activeTaskId)
104
+ mkdirSync(taskDir, { recursive: true })
105
+ appendFileSync(join(taskDir, 'tools.jsonl'), JSON.stringify({
106
+ at: new Date().toISOString(), name, args,
107
+ ...(error ? { error } : { result }),
108
+ }) + '\n')
109
+ }
88
110
  const proxy: AgenticSurface = {
89
111
  ...environment,
90
112
  async call(handle, name, args) {
91
- const res = await environment.call(handle, name, args)
113
+ let res: string
114
+ try {
115
+ res = await environment.call(handle, name, args)
116
+ } catch (error) {
117
+ recordTool(name, args, undefined, error instanceof Error ? error.message : String(error))
118
+ throw error
119
+ }
120
+ recordTool(name, args, res)
92
121
  // Count tool usage per workspace so we can SEE whether the agent ever edited, and whether
93
122
  // edits succeeded or bounced off old_string matching. handle.id keys the workspace.
94
123
  const st = toolStats.get(handle.id) ?? { list: 0, read: 0, edit_ok: 0, edit_fail: 0, run: 0, run_err: 0 }
@@ -161,6 +190,7 @@ async function main(): Promise<void> {
161
190
 
162
191
  let anyResolved = 0
163
192
  for (const task of taskList) {
193
+ activeTaskId = task.id
164
194
  const t0 = Date.now()
165
195
  const r = await runAgentic({
166
196
  surface: proxy,
@@ -171,6 +201,13 @@ async function main(): Promise<void> {
171
201
  workerProfile,
172
202
  analystProfile,
173
203
  budget,
204
+ ...(artifactDir ? { hooks: { onEvent(event: unknown) {
205
+ const taskDir = join(artifactDir, activeTaskId)
206
+ mkdirSync(taskDir, { recursive: true })
207
+ appendFileSync(join(taskDir, 'runtime-events.jsonl'), JSON.stringify(event, (_key, value) =>
208
+ value instanceof Error ? { name: value.name, message: value.message, stack: value.stack } : value,
209
+ ) + '\n')
210
+ } } } : {}),
174
211
  })
175
212
  const rec = captured.get(task.id)
176
213
  const st = toolStats.get(task.id)
@@ -186,6 +223,38 @@ async function main(): Promise<void> {
186
223
  console.log(` patch APPLIED (git apply --check on clean base): ${rec?.applied ? 'YES' : 'NO'}${rec?.applyErr ? ` (${rec.applyErr})` : ''}`)
187
224
  console.log(` swebench judge RESOLVED: ${resolved ? '1' : '0'}`)
188
225
  if (rec?.score?.detail) console.log(` judge report: ${rec.score.detail.slice(0, 400)}`)
226
+ if (artifactDir) {
227
+ const taskDir = join(artifactDir, task.id)
228
+ mkdirSync(taskDir, { recursive: true })
229
+ writeFileSync(join(taskDir, 'patch.diff'), rec?.patch ?? '')
230
+ writeFileSync(join(taskDir, 'summary.json'), JSON.stringify({
231
+ benchmark: 'SWE-bench Verified',
232
+ taskId: task.id,
233
+ modelRequested: model,
234
+ prompt: task.userPrompt,
235
+ turnsLimit: innerTurns,
236
+ maxTokens,
237
+ budget,
238
+ enableRun,
239
+ shots: r.shots,
240
+ completions: r.completions,
241
+ tokens: r.tokens,
242
+ tokensKnown: r.tokensKnown,
243
+ billedCostUsd: null,
244
+ reportedUsd: r.usd,
245
+ reportedUsdKnown: r.usdKnown,
246
+ progression: r.progression,
247
+ runScore: r.score,
248
+ runResolved: r.resolved,
249
+ wallMs: Date.now() - t0,
250
+ toolStats: st ?? null,
251
+ patchBytes,
252
+ patchLines,
253
+ files,
254
+ patchApplies: rec?.applied ?? false,
255
+ score: rec?.score ?? null,
256
+ }, null, 2) + '\n')
257
+ }
189
258
  }
190
259
  console.log(`\n>>> resolved ${anyResolved}/${taskList.length}`)
191
260
  }
@@ -58,7 +58,7 @@ const PROFILE = agentProfileSchema.parse({
58
58
  systemPrompt:
59
59
  'Solve the Terminal-Bench task completely in the provided container and verify the result before finishing.',
60
60
  },
61
- permission: {
61
+ permissions: {
62
62
  edit: 'allow',
63
63
  bash: 'allow',
64
64
  webfetch: 'allow',
@@ -360,14 +360,9 @@ async function captureRunRecord(
360
360
  await appendRunRecord(CORPUS, record)
361
361
  }
362
362
 
363
- /**
364
- * Serially `docker compose build` each task image before the concurrent fan-out.
365
- * tb builds per-task images on first use; building two cold images concurrently
366
- * contends on the shared docker build backend and can return nonzero. Warming the
367
- * cache serially makes the parallel rounds hit a warm cache and never race.
368
- * Dataset-agnostic: derives the compose path from the tb cache layout
369
- * (<cache>/<name>/<version>/<task>/docker-compose.yaml).
370
- */
363
+ /** Warm each task image serially before the concurrent fan-out. Terminal-Bench
364
+ * supplies compose variables only inside its own runner, so build the task's
365
+ * Dockerfile directly and let its later compose build reuse the cached layers. */
371
366
  async function prebuildImages(taskIds: string[]): Promise<void> {
372
367
  const [name, version] = DATASET.split('==')
373
368
  if (!name || !version) {
@@ -376,18 +371,19 @@ async function prebuildImages(taskIds: string[]): Promise<void> {
376
371
  }
377
372
  const cacheRoot = join(homedir(), '.cache', 'terminal-bench', name, version)
378
373
  for (const taskId of taskIds) {
379
- const composePath = join(cacheRoot, taskId, 'docker-compose.yaml')
374
+ const taskDir = join(cacheRoot, taskId)
375
+ const dockerfile = join(taskDir, 'Dockerfile')
380
376
  try {
381
- await stat(composePath)
377
+ await stat(dockerfile)
382
378
  } catch {
383
- console.log(` prebuild: no compose at ${composePath}; tb will materialize ${taskId} on first run`)
379
+ console.log(` prebuild: no Dockerfile at ${dockerfile}; tb will materialize ${taskId} on first run`)
384
380
  continue
385
381
  }
386
382
  process.stdout.write(` prebuild: ${taskId} … `)
387
383
  await new Promise<void>((resolve, reject) => {
388
384
  execFile(
389
385
  'docker',
390
- ['compose', '-p', `tbprebuild-${taskId}`, '-f', composePath, 'build'],
386
+ ['build', '--tag', `tbprebuild-${taskId}`.toLowerCase().replace(/[^a-z0-9_.-]/g, '-'), '--file', dockerfile, taskDir],
391
387
  { maxBuffer: 1024 * 1024 * 256, timeout: ROUND_CAP_MS },
392
388
  (err) => (err ? reject(new Error(`prebuild ${taskId} failed: ${err.message}`)) : resolve()),
393
389
  )
@@ -202,7 +202,7 @@ class OpenCodeRouterAgent(OpenCodeAgent):
202
202
  "$schema": "https://opencode.ai/config.json",
203
203
  # Headless benchmark runs cannot answer interactive permission prompts.
204
204
  # Keep this identical for raw and supervisor arms.
205
- "permission": self._profile.get("permission", {}),
205
+ "permission": self._profile.get("permissions", {}),
206
206
  "provider": {
207
207
  self._provider: {
208
208
  "npm": "@ai-sdk/openai-compatible",