@tangle-network/agent-bench 0.13.11 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,20 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.14.0
4
+
5
+ Require Eval 0.201 through 0.203, Interface 2.15, Knowledge 18, and Sandbox 0.58.4 through the shared catalog.
6
+ Consume Runtime 0.287 through the workspace dependency.
7
+
8
+ ## 0.13.13
9
+
10
+ Require Knowledge 17.1.6 and Eval `>=0.191.0 <0.194.0` through the shared dependency catalog.
11
+ Benchmark behavior is unchanged.
12
+
13
+ ## 0.13.12
14
+
15
+ Accept Eval 0.185 and 0.186 and require Knowledge 17.1.2 through the shared dependency catalog.
16
+ Benchmark behavior is unchanged.
17
+
3
18
  ## 0.13.11
4
19
 
5
20
  Accept Eval 0.184 and require Knowledge 17.1.1 through the shared dependency catalog.
package/HARNESS.md CHANGED
@@ -28,6 +28,8 @@ Use these labels literally. Do not promote one level into another in prose.
28
28
 
29
29
  A localization score, output-shape check, LLM quality judge, or toy deterministic reward can be useful for development. None is a substitute for the benchmark's outcome evaluator.
30
30
 
31
+ The [September 24 public benchmark proof](evidence/public-agent-benchmarks-20260924/README.md) retains three officially graded agent tasks, traces, and validity limits across Terminal-Bench and SWE-bench Verified.
32
+
31
33
  ## Supported commands
32
34
 
33
35
  ### Package and integration contracts
@@ -116,9 +118,9 @@ cd bench
116
118
  pnpm tsx src/swe-self-improve.mts
117
119
  ```
118
120
 
119
- This fixture uses `runStrategyEvolution` with SWE-bench tasks and a frozen holdout.
120
- It does not exercise `improve`, and it deletes its temporary run directory on exit.
121
- It therefore cannot provide retained improvement or lineage evidence.
121
+ This fixture uses `runStrategyEvolution` with SWE-bench tasks split into train, selection and sealed test slices.
122
+ It does not exercise `improve`.
123
+ It keeps its run directory, whose search ledger is the checkpoint and the lineage record of one search.
122
124
  Use `examples/improve` for the maintained offline API fixture.
123
125
  Use the consuming labs for registered learning campaigns with retained execution and comparison evidence.
124
126
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-bench",
3
- "version": "0.13.11",
3
+ "version": "0.14.0",
4
4
  "type": "module",
5
5
  "description": "Benchmark adapters and execution for agent-runtime across coding, tool-use, RAG, memory, browser, and terminal tasks.",
6
6
  "repository": {
@@ -25,11 +25,11 @@
25
25
  }
26
26
  },
27
27
  "dependencies": {
28
- "@tangle-network/agent-eval": ">=0.183.0 <0.185.0",
29
- "@tangle-network/agent-interface": "^2.11.0",
30
- "@tangle-network/agent-knowledge": "^17.1.1",
31
- "@tangle-network/sandbox": ">=0.36.4 <0.48.0",
32
- "@tangle-network/agent-runtime": "^0.256.0"
28
+ "@tangle-network/agent-eval": ">=0.201.0 <0.204.0",
29
+ "@tangle-network/agent-interface": "^2.15.0",
30
+ "@tangle-network/agent-knowledge": "^18.0.0",
31
+ "@tangle-network/sandbox": ">=0.58.4 <0.59.0",
32
+ "@tangle-network/agent-runtime": "^0.287.0"
33
33
  },
34
34
  "devDependencies": {
35
35
  "@arethetypeswrong/cli": "0.18.5",
package/src/selector.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  /**
2
- * Deployable, non-oracle selector (docs/roadmap-rsi.md Phase 1).
2
+ * Deployable, non-oracle selector.
3
3
  *
4
4
  * best-of-N only pays if you can pick the good attempt WITHOUT the judge. Today
5
5
  * the loop's winner is judge-selected (verdict.score) — an oracle upper bound, not
@@ -277,8 +277,9 @@ async function resolveInstanceImage(ws: Ws, expected?: SweImageIdentity): Promis
277
277
  }
278
278
 
279
279
  /** Build the SWE-bench Environment + a DISJOINT-slice task supplier over the Verified split. The
280
- * supplier keys tasks by dataset offset so `runStrategyEvolution`'s train [0,trainN) and holdout
281
- * [trainN+off,…) never overlap. Verified is loaded once; instances carry their repo/base_commit. */
280
+ * supplier keys tasks by dataset offset, so consecutive slices (train, selection and test for
281
+ * `runStrategyEvolution`) never overlap. Verified is loaded once; instances carry their
282
+ * repo/base_commit. */
282
283
  export async function createSweBenchEnvironment(
283
284
  poolN = 80,
284
285
  opts: {
@@ -16,7 +16,7 @@
16
16
  * node_modules/.bin/tsx bench/src/swe-local-proof.mts
17
17
  */
18
18
  import { execFile } from 'node:child_process'
19
- import { mkdtempSync, rmSync, writeFileSync } from 'node:fs'
19
+ import { appendFileSync, mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'
20
20
  import { tmpdir } from 'node:os'
21
21
  import { join } from 'node:path'
22
22
  import { promisify } from 'node:util'
@@ -40,12 +40,24 @@ async function main(): Promise<void> {
40
40
  // WITH-TOOLS arm: RUN_TOOL=1 exposes the jailed `run` tool + the run-aware prompt. Default OFF ⇒
41
41
  // the read/edit-only baseline (glm-5.2 7/12) unchanged.
42
42
  const enableRun = ['1', 'true', 'yes'].includes((process.env.RUN_TOOL ?? '').toLowerCase())
43
+ const artifactDir = process.env.RUN_ARTIFACT_DIR
44
+ if (artifactDir) mkdirSync(artifactDir, { recursive: true })
43
45
 
44
46
  console.log(`═══ SWE-bench Verified — SEE-able LOCAL proof (no tangle sandbox) ═══`)
45
47
  console.log(`model=${model} ids=${ids.join(',')} innerTurns=${innerTurns} maxTokens=${maxTokens} budget=${budget} runTool=${enableRun}`)
46
48
  console.log(`router=${routerBaseUrl}`)
47
49
 
48
- const { environment, tasks, adapter } = await createSweBenchEnvironment(ids.length, { ids, enableRun })
50
+ const { environment, tasks, adapter } = await createSweBenchEnvironment(ids.length, {
51
+ ids,
52
+ enableRun,
53
+ ...(artifactDir ? {
54
+ adapterOptions: {
55
+ captureEvaluatorArtifacts: ({ taskId, attemptSequence }) => ({
56
+ destination: join(artifactDir, taskId, `judge-${attemptSequence}`),
57
+ }),
58
+ },
59
+ } : {}),
60
+ })
49
61
  const workerProfile = withBenchProfile(
50
62
  {
51
63
  name: 'swe-local-proof-worker',
@@ -85,10 +97,27 @@ async function main(): Promise<void> {
85
97
  const captured = new Map<string, Rec>()
86
98
  const judged = new Map<string, BenchScore>()
87
99
  const toolStats = new Map<string, { list: number; read: number; edit_ok: number; edit_fail: number; run: number; run_err: number }>()
100
+ let activeTaskId = ''
101
+ const recordTool = (name: string, args: unknown, result: unknown, error?: string) => {
102
+ if (!artifactDir || !activeTaskId) return
103
+ const taskDir = join(artifactDir, activeTaskId)
104
+ mkdirSync(taskDir, { recursive: true })
105
+ appendFileSync(join(taskDir, 'tools.jsonl'), JSON.stringify({
106
+ at: new Date().toISOString(), name, args,
107
+ ...(error ? { error } : { result }),
108
+ }) + '\n')
109
+ }
88
110
  const proxy: AgenticSurface = {
89
111
  ...environment,
90
112
  async call(handle, name, args) {
91
- const res = await environment.call(handle, name, args)
113
+ let res: string
114
+ try {
115
+ res = await environment.call(handle, name, args)
116
+ } catch (error) {
117
+ recordTool(name, args, undefined, error instanceof Error ? error.message : String(error))
118
+ throw error
119
+ }
120
+ recordTool(name, args, res)
92
121
  // Count tool usage per workspace so we can SEE whether the agent ever edited, and whether
93
122
  // edits succeeded or bounced off old_string matching. handle.id keys the workspace.
94
123
  const st = toolStats.get(handle.id) ?? { list: 0, read: 0, edit_ok: 0, edit_fail: 0, run: 0, run_err: 0 }
@@ -161,6 +190,7 @@ async function main(): Promise<void> {
161
190
 
162
191
  let anyResolved = 0
163
192
  for (const task of taskList) {
193
+ activeTaskId = task.id
164
194
  const t0 = Date.now()
165
195
  const r = await runAgentic({
166
196
  surface: proxy,
@@ -171,6 +201,13 @@ async function main(): Promise<void> {
171
201
  workerProfile,
172
202
  analystProfile,
173
203
  budget,
204
+ ...(artifactDir ? { hooks: { onEvent(event: unknown) {
205
+ const taskDir = join(artifactDir, activeTaskId)
206
+ mkdirSync(taskDir, { recursive: true })
207
+ appendFileSync(join(taskDir, 'runtime-events.jsonl'), JSON.stringify(event, (_key, value) =>
208
+ value instanceof Error ? { name: value.name, message: value.message, stack: value.stack } : value,
209
+ ) + '\n')
210
+ } } } : {}),
174
211
  })
175
212
  const rec = captured.get(task.id)
176
213
  const st = toolStats.get(task.id)
@@ -186,6 +223,38 @@ async function main(): Promise<void> {
186
223
  console.log(` patch APPLIED (git apply --check on clean base): ${rec?.applied ? 'YES' : 'NO'}${rec?.applyErr ? ` (${rec.applyErr})` : ''}`)
187
224
  console.log(` swebench judge RESOLVED: ${resolved ? '1' : '0'}`)
188
225
  if (rec?.score?.detail) console.log(` judge report: ${rec.score.detail.slice(0, 400)}`)
226
+ if (artifactDir) {
227
+ const taskDir = join(artifactDir, task.id)
228
+ mkdirSync(taskDir, { recursive: true })
229
+ writeFileSync(join(taskDir, 'patch.diff'), rec?.patch ?? '')
230
+ writeFileSync(join(taskDir, 'summary.json'), JSON.stringify({
231
+ benchmark: 'SWE-bench Verified',
232
+ taskId: task.id,
233
+ modelRequested: model,
234
+ prompt: task.userPrompt,
235
+ turnsLimit: innerTurns,
236
+ maxTokens,
237
+ budget,
238
+ enableRun,
239
+ shots: r.shots,
240
+ completions: r.completions,
241
+ tokens: r.tokens,
242
+ tokensKnown: r.tokensKnown,
243
+ billedCostUsd: null,
244
+ reportedUsd: r.usd,
245
+ reportedUsdKnown: r.usdKnown,
246
+ progression: r.progression,
247
+ runScore: r.score,
248
+ runResolved: r.resolved,
249
+ wallMs: Date.now() - t0,
250
+ toolStats: st ?? null,
251
+ patchBytes,
252
+ patchLines,
253
+ files,
254
+ patchApplies: rec?.applied ?? false,
255
+ score: rec?.score ?? null,
256
+ }, null, 2) + '\n')
257
+ }
189
258
  }
190
259
  console.log(`\n>>> resolved ${anyResolved}/${taskList.length}`)
191
260
  }
@@ -1,20 +1,22 @@
1
1
  /**
2
- * SWE-bench self-improvement — the PROPER, no-cheating run: a frontier worker over the SWE-bench
3
- * `Environment`, with `runStrategyEvolution` enforcing the train→freeze→holdout split (the substrate
4
- * draws a disjoint holdout slice and gates once — adaptive reuse is impossible). CONTAMINATION CAVEAT
5
- * applies (public fixes may be memorized) — reported, never claimed clean.
2
+ * SWE-bench self-improvement: a frontier worker over the SWE-bench `Environment`, with
3
+ * `runStrategyEvolution` searching strategies on disjoint train, selection and test slices. The
4
+ * author reads train results only, strategies are ranked on the private selection slice, and the
5
+ * claim runs once on the sealed test slice. CONTAMINATION CAVEAT applies (public fixes may be
6
+ * memorized) — reported, never claimed clean.
6
7
  *
7
8
  * CALIBRATE first (cost gate): TANGLE_API_KEY=… CALIBRATE=1 N=3 tsx bench/src/swe-self-improve.mts
8
- * Full run: TANGLE_API_KEY=… TRAIN_N=6 HOLDOUT_N=8 GENERATIONS=2 tsx bench/src/swe-self-improve.mts
9
+ * Full run: TANGLE_API_KEY=… TRAIN_N=4 SELECTION_N=12 TEST_N=12 tsx bench/src/swe-self-improve.mts
10
+ *
11
+ * The run directory is kept (`OUT_DIR`, default `.swe-run`): its ledger is the checkpoint, so the
12
+ * same command continues an interrupted search.
9
13
  */
10
- import { mkdtempSync, rmSync } from 'node:fs'
11
14
  import { join } from 'node:path'
12
- import type { AgentProfile } from '@tangle-network/agent-interface'
15
+ import { type AgentProfile, canonicalCandidateDigest } from '@tangle-network/agent-interface'
13
16
  import {
14
17
  refine,
15
18
  runAgentic,
16
19
  runStrategyEvolution,
17
- sample,
18
20
  strategyAuthorSystemPrompt,
19
21
  } from '@tangle-network/agent-runtime/kernel'
20
22
  import { createSweBenchEnvironment } from './swe-bench-env'
@@ -72,46 +74,52 @@ async function main(): Promise<void> {
72
74
  return
73
75
  }
74
76
 
75
- const report = await (async () => {
76
- const outDir = mkdtempSync(join(process.cwd(), '.swe-run-'))
77
- try {
78
- return await runStrategyEvolution({
79
- environment,
80
- tasks,
81
- trainN: Number(process.env.TRAIN_N ?? 6),
82
- holdoutN: Number(process.env.HOLDOUT_N ?? 8),
83
- worker: { routerBaseUrl, routerKey, workerProfile },
84
- author: {
85
- profile: authorProfile(authorModel, 'swe-strategy-author'),
86
- executor: { backend: 'router', routerBaseUrl, routerKey },
87
- fallbackProfile: authorProfile(
88
- process.env.AUTHOR_FALLBACK ?? 'deepseek-v4-flash',
89
- 'swe-strategy-author-fallback',
90
- ),
91
- },
92
- baselines: [sample, refine],
93
- budget: Number(process.env.BUDGET ?? 2),
94
- generations: Number(process.env.GENERATIONS ?? 2),
95
- populationSize: Number(process.env.POP ?? 2),
96
- outDir,
97
- })
98
- } finally {
99
- rmSync(outDir, { recursive: true, force: true })
100
- }
101
- })()
77
+ const trainN = Number(process.env.TRAIN_N ?? 4)
78
+ const selectionN = Number(process.env.SELECTION_N ?? 12)
79
+ const testN = Number(process.env.TEST_N ?? 12)
80
+ const all = await tasks(0, trainN + selectionN + testN)
81
+ const report = await runStrategyEvolution({
82
+ environment,
83
+ train: all.slice(0, trainN),
84
+ selection: all.slice(trainN, trainN + selectionN),
85
+ test: all.slice(trainN + selectionN),
86
+ claim: {
87
+ use: 'comparison',
88
+ population: { id: 'swe-bench-verified', description: 'SWE-bench Verified instances' },
89
+ samplingFrame: 'consecutive SWE-bench Verified instances from the loaded pool',
90
+ independentUnit: 'id',
91
+ generalization: 'new-units',
92
+ minimumEffect: Number(process.env.MIN_EFFECT ?? 0.15),
93
+ },
94
+ executionRef: canonicalCandidateDigest({
95
+ environment: 'bench/src/swe-bench-env.ts',
96
+ harness: 'swebench-docker',
97
+ innerTurns,
98
+ }),
99
+ worker: { routerBaseUrl, routerKey, workerProfile },
100
+ author: {
101
+ profile: authorProfile(authorModel, 'swe-strategy-author'),
102
+ executor: { backend: 'router', routerBaseUrl, routerKey },
103
+ fallbackProfile: authorProfile(
104
+ process.env.AUTHOR_FALLBACK ?? 'deepseek-v4-flash',
105
+ 'swe-strategy-author-fallback',
106
+ ),
107
+ },
108
+ root: refine,
109
+ budget: Number(process.env.BUDGET ?? 2),
110
+ maxExpansions: Number(process.env.EXPANSIONS ?? 4),
111
+ outDir: join(process.cwd(), process.env.OUT_DIR ?? '.swe-run'),
112
+ })
102
113
 
103
- const v = report.verdict
104
- console.log('\n═══ SWE-bench SELF-IMPROVEMENT — certified on a FROZEN holdout (CONTAMINATION-flagged) ═══')
105
- console.log(`worker=${workerModel} author=${authorModel}`)
106
- console.log(`gen0 champion: ${report.gen0Champion.name}`)
107
- console.log(`final champion: ${report.finalChampion.name}`)
108
- console.log(`PROMOTED: ${v.promoted} (${v.reason})`)
109
- console.log(`held-out lift: mean ${v.lift.mean.toFixed(3)} 95% CI [${v.lift.low.toFixed(3)}, ${v.lift.high.toFixed(3)}] n=${v.n}`)
110
- console.log(
111
- v.promoted
112
- ? '\n>>> The search taught the agent a strategy that resolves MORE real bugs it never trained on, beyond luck. (Report the contamination caveat: public fixes may be memorized.)'
113
- : '\n>>> No promotion: the evolved strategy did not beat gen0 on the fresh holdout beyond noise (honest null).',
114
- )
114
+ const shipped = report.claim.finalists.find((f) => f.nodeId === report.claim.selected)?.test
115
+ console.log('\n═══ SWE-bench SELF-IMPROVEMENT — claimed on a SEALED test slice (CONTAMINATION-flagged) ═══')
116
+ console.log(`worker=${workerModel} author=${authorModel} ledger=${report.ledger}`)
117
+ console.log(`strategies: ${report.strategies.map((s) => `${s.name} (${s.status})`).join(', ')}`)
118
+ console.log(`selected: ${report.selected.name}`)
119
+ console.log(`DECISION: ${report.decision} (${report.reason})`)
120
+ if (shipped) {
121
+ console.log(`test lift: ${shipped.delta.toFixed(3)} [${shipped.interval[0].toFixed(3)}, ${shipped.interval[1].toFixed(3)}] n=${shipped.pairs}`)
122
+ }
115
123
  }
116
124
 
117
125
  main().catch((e) => {
@@ -58,7 +58,7 @@ const PROFILE = agentProfileSchema.parse({
58
58
  systemPrompt:
59
59
  'Solve the Terminal-Bench task completely in the provided container and verify the result before finishing.',
60
60
  },
61
- permission: {
61
+ permissions: {
62
62
  edit: 'allow',
63
63
  bash: 'allow',
64
64
  webfetch: 'allow',
@@ -360,14 +360,9 @@ async function captureRunRecord(
360
360
  await appendRunRecord(CORPUS, record)
361
361
  }
362
362
 
363
- /**
364
- * Serially `docker compose build` each task image before the concurrent fan-out.
365
- * tb builds per-task images on first use; building two cold images concurrently
366
- * contends on the shared docker build backend and can return nonzero. Warming the
367
- * cache serially makes the parallel rounds hit a warm cache and never race.
368
- * Dataset-agnostic: derives the compose path from the tb cache layout
369
- * (<cache>/<name>/<version>/<task>/docker-compose.yaml).
370
- */
363
+ /** Warm each task image serially before the concurrent fan-out. Terminal-Bench
364
+ * supplies compose variables only inside its own runner, so build the task's
365
+ * Dockerfile directly and let its later compose build reuse the cached layers. */
371
366
  async function prebuildImages(taskIds: string[]): Promise<void> {
372
367
  const [name, version] = DATASET.split('==')
373
368
  if (!name || !version) {
@@ -376,18 +371,19 @@ async function prebuildImages(taskIds: string[]): Promise<void> {
376
371
  }
377
372
  const cacheRoot = join(homedir(), '.cache', 'terminal-bench', name, version)
378
373
  for (const taskId of taskIds) {
379
- const composePath = join(cacheRoot, taskId, 'docker-compose.yaml')
374
+ const taskDir = join(cacheRoot, taskId)
375
+ const dockerfile = join(taskDir, 'Dockerfile')
380
376
  try {
381
- await stat(composePath)
377
+ await stat(dockerfile)
382
378
  } catch {
383
- console.log(` prebuild: no compose at ${composePath}; tb will materialize ${taskId} on first run`)
379
+ console.log(` prebuild: no Dockerfile at ${dockerfile}; tb will materialize ${taskId} on first run`)
384
380
  continue
385
381
  }
386
382
  process.stdout.write(` prebuild: ${taskId} … `)
387
383
  await new Promise<void>((resolve, reject) => {
388
384
  execFile(
389
385
  'docker',
390
- ['compose', '-p', `tbprebuild-${taskId}`, '-f', composePath, 'build'],
386
+ ['build', '--tag', `tbprebuild-${taskId}`.toLowerCase().replace(/[^a-z0-9_.-]/g, '-'), '--file', dockerfile, taskDir],
391
387
  { maxBuffer: 1024 * 1024 * 256, timeout: ROUND_CAP_MS },
392
388
  (err) => (err ? reject(new Error(`prebuild ${taskId} failed: ${err.message}`)) : resolve()),
393
389
  )
@@ -202,7 +202,7 @@ class OpenCodeRouterAgent(OpenCodeAgent):
202
202
  "$schema": "https://opencode.ai/config.json",
203
203
  # Headless benchmark runs cannot answer interactive permission prompts.
204
204
  # Keep this identical for raw and supervisor arms.
205
- "permission": self._profile.get("permission", {}),
205
+ "permission": self._profile.get("permissions", {}),
206
206
  "provider": {
207
207
  self._provider: {
208
208
  "npm": "@ai-sdk/openai-compatible",