@tangle-network/agent-bench 0.9.4 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-bench",
3
- "version": "0.9.4",
3
+ "version": "0.11.0",
4
4
  "type": "module",
5
5
  "description": "Benchmark adapters and execution for agent-runtime across coding, tool-use, RAG, memory, browser, and terminal tasks.",
6
6
  "repository": {
@@ -25,11 +25,11 @@
25
25
  }
26
26
  },
27
27
  "dependencies": {
28
- "@tangle-network/agent-eval": ">=0.176.0 <0.177.0",
29
- "@tangle-network/agent-interface": "^2.5.0",
30
- "@tangle-network/agent-knowledge": "^15.0.0",
28
+ "@tangle-network/agent-eval": ">=0.178.0 <0.179.0",
29
+ "@tangle-network/agent-interface": "^2.6.0",
30
+ "@tangle-network/agent-knowledge": "^15.0.1",
31
31
  "@tangle-network/sandbox": ">=0.36.4 <0.38.0",
32
- "@tangle-network/agent-runtime": "^0.202.0"
32
+ "@tangle-network/agent-runtime": "^0.203.2"
33
33
  },
34
34
  "devDependencies": {
35
35
  "@arethetypeswrong/cli": "0.18.5",
@@ -1,5 +1,6 @@
1
1
  import { execFile } from 'node:child_process'
2
- import { access, readdir, readFile } from 'node:fs/promises'
2
+ import { access, readdir, readFile, realpath } from 'node:fs/promises'
3
+ import { tmpdir } from 'node:os'
3
4
  import path from 'node:path'
4
5
  import { fileURLToPath } from 'node:url'
5
6
  import { promisify } from 'node:util'
@@ -64,6 +65,15 @@ export async function run(command, args, env = process.env) {
64
65
  }
65
66
  }
66
67
 
68
+ export async function runPythonTests(python, env = process.env) {
69
+ // Fixture roots must be physical paths; the production boundary rejects symlinked ancestors.
70
+ const physicalTemp = await realpath(env.TMPDIR || tmpdir())
71
+ await run(python, ['-m', 'unittest', 'discover', '-s', 'pier_agents', '-p', '*_test.py'], {
72
+ ...env,
73
+ TMPDIR: physicalTemp,
74
+ })
75
+ }
76
+
67
77
  async function main() {
68
78
  const python = path.join(benchDir, '.venv', 'bin', 'python')
69
79
  try {
@@ -101,7 +111,7 @@ async function main() {
101
111
  await run('npx', ['vitest', 'run', ...vitestTests.map((file) => path.relative(benchDir, file))])
102
112
  }
103
113
 
104
- await run(python, ['-m', 'unittest', 'discover', '-s', 'pier_agents', '-p', '*_test.py'])
114
+ await runPythonTests(python)
105
115
 
106
116
  console.log(
107
117
  `package tests passed: ${tests.length}/${tests.length} TypeScript files (${nodeTests.length} node:test + ${vitestTests.length} vitest) + Pier bridge`,
@@ -1,6 +1,9 @@
1
1
  import assert from 'node:assert/strict'
2
2
  import { test } from 'node:test'
3
- import { packageNodeTestArgs, resolvePackageTestTimeoutMs, run } from './run-package-tests.mjs'
3
+ import { mkdtemp, mkdir, readFile, realpath, rm, symlink, writeFile } from 'node:fs/promises'
4
+ import { tmpdir } from 'node:os'
5
+ import path from 'node:path'
6
+ import { packageNodeTestArgs, resolvePackageTestTimeoutMs, run, runPythonTests } from './run-package-tests.mjs'
4
7
 
5
8
  test('package test concurrency reaches Node without changing selected files', () => {
6
9
  const files = ['src/first.test.mts', 'src/second.test.ts']
@@ -45,3 +48,25 @@ test('package test timeout reaches the child process', async () => {
45
48
  )
46
49
  assert.ok(Date.now() - startedAt < 2_000)
47
50
  })
51
+
52
+ test('Python fixtures use physical temporary paths without relaxing symlink guards', async () => {
53
+ const root = await mkdtemp(path.join(await realpath(tmpdir()), 'bench-python-temp-'))
54
+ try {
55
+ const target = path.join(root, 'physical')
56
+ const linked = path.join(root, 'linked')
57
+ const recorded = path.join(root, 'recorded.json')
58
+ const launcher = path.join(root, 'python-fixture')
59
+ await mkdir(target)
60
+ await symlink(target, linked, 'dir')
61
+ await writeFile(launcher, `#!${process.execPath}
62
+ const fs = require('node:fs'); fs.writeFileSync(process.env.RECORDED, JSON.stringify({ temporary: process.env.TMPDIR, args: process.argv.slice(2) }))
63
+ `, { mode: 0o755 })
64
+ await runPythonTests(launcher, { ...process.env, TMPDIR: linked, RECORDED: recorded })
65
+ assert.deepEqual(JSON.parse(await readFile(recorded, 'utf8')), {
66
+ temporary: target,
67
+ args: ['-m', 'unittest', 'discover', '-s', 'pier_agents', '-p', '*_test.py'],
68
+ })
69
+ } finally {
70
+ await rm(root, { recursive: true, force: true })
71
+ }
72
+ })
@@ -165,6 +165,51 @@ writeFileSync(
165
165
  path.join(consumerDir, 'index.ts'),
166
166
  "import { createSweBenchAdapter, executePreparedPierCandidate, FilePierCandidateTrialController, resolveAdapter, runBenchmarks, runStagedJudge, StagedJudgeError, type BenchmarkAdapter, type JudgeArtifactReceipt, type PierCandidateTrialController, type PierCandidateTrialHandle, type PierDockerConnection, type StagedPierCandidateExecution } from '@tangle-network/agent-bench'\n\nconst adapter: BenchmarkAdapter = resolveAdapter('swe-bench')\nconst captureAdapter: BenchmarkAdapter = createSweBenchAdapter({ captureEvaluatorArtifacts: ({ taskId, attemptSequence }) => ({ destination: `/tmp/${taskId}/${attemptSequence}` }) })\nconst receipt = undefined as JudgeArtifactReceipt | undefined\nconst staged = undefined as StagedPierCandidateExecution | undefined\nconst trial = undefined as PierCandidateTrialHandle | undefined\nconst controller = undefined as PierCandidateTrialController | undefined\nconst dockerConnection = undefined as PierDockerConnection | undefined\nvoid adapter\nvoid captureAdapter\nvoid receipt\nvoid staged\nvoid trial\nvoid controller\nvoid dockerConnection\nvoid executePreparedPierCandidate\nvoid FilePierCandidateTrialController\nvoid runBenchmarks\nvoid runStagedJudge\nvoid StagedJudgeError\n",
167
167
  )
168
+ await writeFile(
169
+ path.join(consumerDir, 'managed-execution.ts'),
170
+ `import assert from 'node:assert/strict'
171
+ import { runBenchmarks, type BenchExecution, type BenchExecutionContext, type BenchPromptResult } from '@tangle-network/agent-bench'
172
+
173
+ const sessions: Array<string | undefined> = []
174
+ let closed = 0
175
+ const execute: BenchExecution = async (context: BenchExecutionContext) => {
176
+ await context.run.start(context.prompt)
177
+ await context.run.resume('corrected')
178
+ }
179
+ const report = await runBenchmarks({
180
+ benchmarks: ['fixture'], cells: [{ label: 'worker', model: 'fixture' }],
181
+ routerBaseUrl: 'unused', routerKey: 'unused', verifyJudge: false, execute,
182
+ resolveAdapter: () => ({
183
+ name: 'fixture', preflight: async () => {},
184
+ loadTasks: async () => [{ id: 'task', prompt: 'initial' }],
185
+ goldArtifact: async () => undefined,
186
+ judge: async (_task, artifact) => ({ resolved: artifact === 'corrected', score: artifact === 'corrected' ? 1 : 0 }),
187
+ }),
188
+ resolveClient: () => ({
189
+ criuStatus: async () => ({ available: false }),
190
+ create: async () => ({
191
+ id: 'packed-fixture',
192
+ async *streamPrompt(prompt: string, options?: { sessionId?: string }) {
193
+ sessions.push(options?.sessionId)
194
+ yield { type: 'llm_call', data: { tokensIn: 2, tokensOut: 1, costUsd: 0.01 } }
195
+ yield { type: 'result', data: { finalText: prompt, success: true, status: 'success' } }
196
+ yield { type: 'done', data: { outcome: { type: 'completed' } } }
197
+ },
198
+ delete: async () => { closed += 1 },
199
+ }),
200
+ }) as never,
201
+ })
202
+ const prompts: readonly BenchPromptResult[] = report.perTask[0]?.prompts ?? []
203
+ assert.equal(report.perTask[0]?.resolved, true)
204
+ assert.equal(prompts.length, 2)
205
+ assert.equal(prompts[1]?.method, 'resume')
206
+ assert.equal(prompts[1]?.prompt, 'corrected')
207
+ assert.equal(sessions[0], sessions[1])
208
+ assert.ok(sessions[0])
209
+ assert.equal(closed, 1)
210
+ assert.deepEqual(report.perTask[0]?.usage, { input: 4, output: 2, costUsd: 0.02 })
211
+ `,
212
+ )
168
213
  await writeFile(
169
214
  path.join(consumerDir, 'index.mjs'),
170
215
  "import { resolveAdapter, runBenchmarks } from '@tangle-network/agent-bench'\nimport { ADAPTERS } from '@tangle-network/agent-bench/adapters'\nimport { createCragAdapter } from '@tangle-network/agent-bench/benchmarks/crag'\n\nif (typeof resolveAdapter !== 'function' || typeof runBenchmarks !== 'function') throw new Error('root exports are not executable')\nif (typeof ADAPTERS !== 'object' || typeof createCragAdapter !== 'function') throw new Error('subpath exports are not executable')\nif (resolveAdapter('crag').name !== 'crag') throw new Error('compiled adapter registry returned the wrong adapter')\nprocess.env.TOOLLM_FIXTURES = '1'\nconst toolLlmTasks = await resolveAdapter('toollm').loadTasks({ limit: 1 })\nif (toolLlmTasks.length !== 1 || toolLlmTasks[0]?.id !== '1') throw new Error('packed ToolLLM fixture loading failed')\n",
@@ -189,7 +234,7 @@ if (!score.resolved || score.score !== 1) {
189
234
  `${JSON.stringify(
190
235
  {
191
236
  compilerOptions: publicTsconfig.compilerOptions,
192
- files: ['index.ts'],
237
+ files: ['index.ts', 'managed-execution.ts'],
193
238
  },
194
239
  null,
195
240
  2,
@@ -255,6 +300,7 @@ for name in sorted(expected):
255
300
  throw new Error(`expected TypeScript ${TYPESCRIPT_6}, received ${typescript6.stdout.trim()}`)
256
301
  }
257
302
  await run('npm', ['exec', '--', 'tsx', 'index.ts'], consumerDir)
303
+ await run('npm', ['exec', '--', 'tsx', 'managed-execution.ts'], consumerDir)
258
304
  const installedPackage = path.join(consumerDir, 'node_modules', '@tangle-network', 'agent-bench')
259
305
  const prepared = await run(
260
306
  'npm',
package/src/index.ts CHANGED
@@ -64,6 +64,9 @@ export {
64
64
  runBenchmarks,
65
65
  printBenchmarksReport,
66
66
  type BenchCell,
67
+ type BenchExecution,
68
+ type BenchExecutionContext,
69
+ type BenchPromptResult,
67
70
  type BenchShot,
68
71
  type BenchShotResult,
69
72
  type BenchCellTaskResult,
@@ -5,7 +5,7 @@
5
5
  */
6
6
  import assert from 'node:assert/strict'
7
7
  import type { BenchmarkAdapter, BenchScore, BenchTask } from './benchmarks/types'
8
- import { runBenchmarks, type BenchShot } from './run-benchmarks'
8
+ import { runBenchmarks, type BenchShot, type BenchExecution, type BenchExecutionContext } from './run-benchmarks'
9
9
 
10
10
  function stubAdapter(name: string, n: number): BenchmarkAdapter {
11
11
  const tasks: BenchTask[] = Array.from({ length: n }, (_, i) => ({
@@ -47,6 +47,7 @@ const shot: BenchShot = async ({ adapter, task, cell }) => {
47
47
  }
48
48
 
49
49
  async function main(): Promise<void> {
50
+ await managedExecutionProof()
50
51
  // Matrix: 2 benchmarks × 3 cells × 4 tasks = 24 shots.
51
52
  const report = await runBenchmarks({
52
53
  benchmarks: ['alpha', 'beta'],
@@ -77,6 +78,22 @@ async function main(): Promise<void> {
77
78
  assert.equal(judgeFailure.perTask[0]?.ok, false)
78
79
  assert.equal(judgeFailure.perTask[0]?.artifact, 'PATCH', 'a judge outage retains the completed agent artifact')
79
80
  assert.equal(judgeFailure.perTask[0]?.usage?.costUsd, 0.04, 'a judge outage retains already incurred usage')
81
+ assert.equal(judgeFailure.perTask[0]?.measurement, 'unavailable')
82
+ assert.deepEqual(judgeFailure.perTask[0]?.execution, { phase: 'started', terminalOutcome: 'succeeded' })
83
+
84
+ const failedExecution = await runBenchmarks({
85
+ benchmarks: ['alpha'], cells: [{ label: 'paid-failure', model: 'm' }],
86
+ routerBaseUrl: 'x', routerKey: 'x', n: 1, resolveAdapter: resolveStub,
87
+ runShot: async () => ({
88
+ artifact: '', ok: false, artifactAvailable: true,
89
+ execution: { phase: 'started', terminalOutcome: 'failed' },
90
+ usage: { input: 23, output: 7, costUsd: 0.04 },
91
+ }),
92
+ })
93
+ assert.equal(failedExecution.perTask[0]?.measurement, 'available')
94
+ assert.equal(failedExecution.perTask[0]?.resolved, false)
95
+ assert.equal(failedExecution.rows[0]?.errored, 0, 'a measured paid failure stays in the comparison')
96
+ assert.equal(failedExecution.perTask[0]?.usage?.input, 23)
80
97
 
81
98
  const controller = new AbortController()
82
99
  let started = 0
@@ -93,6 +110,7 @@ async function main(): Promise<void> {
93
110
  })
94
111
  assert.equal(started, 1, 'cancellation prevents every queued model call')
95
112
  assert.equal(cancelled.perTask.length, 4, 'cancelled work remains visible')
113
+ assert.equal(cancelled.perTask.filter((row) => row.execution?.phase === 'not-started').length, 3)
96
114
  assert.equal(cancelled.perTask[0]?.usage?.costUsd, 0.01)
97
115
 
98
116
  const row = (b: string, c: string) => report.rows.find((r) => r.benchmark === b && r.cell === c)!
@@ -159,6 +177,20 @@ async function main(): Promise<void> {
159
177
  input: 11, output: 3, costUsd: 0.02, tokensKnown: false, usdKnown: false,
160
178
  }, 'an unreported retry preserves the measured floor without claiming complete accounting')
161
179
 
180
+ const measuredFailedRetry = await runBenchmarks({
181
+ benchmarks: ['alpha'], cells: [{ label: 'failed-retries', model: 'm' }],
182
+ routerBaseUrl: 'x', routerKey: 'x', resolveAdapter: resolveStub, n: 1, loopAttempts: 2,
183
+ runShot: async ({ attempt }) => ({
184
+ artifact: attempt === 1 ? '' : 'WRONG', ok: false, artifactAvailable: attempt === 2,
185
+ execution: { phase: 'started', terminalOutcome: 'failed' },
186
+ usage: { input: 11, output: 3, costUsd: 0.02 },
187
+ }),
188
+ })
189
+ assert.equal(measuredFailedRetry.perTask[0]?.artifact, 'WRONG')
190
+ assert.equal(measuredFailedRetry.perTask[0]?.measurement, 'available')
191
+ assert.equal(measuredFailedRetry.rows[0]?.errored, 0, 'a measured failed retry stays in the comparison')
192
+ assert.deepEqual(measuredFailedRetry.perTask[0]?.usage, { input: 22, output: 6, costUsd: 0.04 })
193
+
162
194
  let validAttempts = 0
163
195
  const validRetry = await runBenchmarks({
164
196
  benchmarks: ['alpha'], cells: [{ label: 'retrying', model: 'm' }],
@@ -212,6 +244,10 @@ async function main(): Promise<void> {
212
244
  const order: string[] = []
213
245
  let createdOptions: unknown
214
246
  let controlCredential: string | undefined
247
+ let completed = true
248
+ let text = 'fallback text'
249
+ let streamThrows = false
250
+ let setupFails = false
215
251
  const fakeClient = {
216
252
  async create(options: unknown) {
217
253
  createdOptions = options
@@ -219,13 +255,14 @@ async function main(): Promise<void> {
219
255
  id: 'box-default-shot',
220
256
  async exec(command: string, options?: { sessionId?: string }) {
221
257
  order.push(`exec:${command}:streams=${order.filter((x) => x.startsWith('stream:')).length}:session=${options?.sessionId ? 'yes' : 'no'}`)
222
- return { exitCode: 0, stdout: command === 'extract-patch' ? 'PATCH' : '', stderr: '' }
258
+ return { exitCode: setupFails && command === 'setup-repo' ? 1 : 0, stdout: command === 'extract-patch' ? 'PATCH' : '', stderr: '' }
223
259
  },
224
260
  async *streamPrompt(_prompt: string, options?: { sessionId?: string }) {
225
261
  order.push(`stream:session=${options?.sessionId ? 'yes' : 'no'}`)
226
262
  yield { type: 'llm_call', data: { tokensIn: 23, tokensOut: 7, costUsd: 0.04 } }
227
- yield { type: 'result', data: { finalText: 'fallback text' } }
228
- yield { type: 'done', data: { outcome: { type: 'completed' } } }
263
+ if (streamThrows) throw new Error('stream disconnected')
264
+ yield { type: 'result', data: { finalText: text, success: completed, status: completed ? 'success' : 'failed' } }
265
+ yield { type: 'done', data: { outcome: { type: completed ? 'completed' : 'failed' } } }
229
266
  },
230
267
  async delete() {
231
268
  order.push('delete')
@@ -260,6 +297,8 @@ async function main(): Promise<void> {
260
297
  assert.equal(boxy.rows[0]!.resolveRate, 1, 'boxExtract artifact is judged instead of fallback text')
261
298
  assert.equal(boxy.perTask[0]?.artifact, 'PATCH')
262
299
  assert.deepEqual(boxy.perTask[0]?.usage, { input: 23, output: 7, costUsd: 0.04 })
300
+ assert.deepEqual(boxy.perTask[0]?.execution, { phase: 'started', terminalOutcome: 'succeeded' })
301
+ assert.equal(boxy.perTask[0]?.measurement, 'available')
263
302
  assert.equal(controlCredential, 'sandbox-control-token', 'inference grant never authorizes sandbox control')
264
303
  assert.equal((createdOptions as { backend: { model: { apiKey: string } } }).backend.model.apiKey, 'model-grant-token')
265
304
  assert.deepEqual(
@@ -267,6 +306,49 @@ async function main(): Promise<void> {
267
306
  ['exec:setup-repo:streams=0:session=yes', 'stream:session=yes', 'exec:extract-patch:streams=1:session=yes'],
268
307
  'setup runs before the prompt stream, extract runs after the prompt stream, both in the same session',
269
308
  )
309
+ for (const failure of ['stream', 'parser'] as const) {
310
+ streamThrows = failure === 'stream'
311
+ const failed = await runBenchmarks({
312
+ benchmarks: ['boxy'], cells: [{ label: failure, model: 'm', backend: 'sandbox' }],
313
+ routerBaseUrl: 'x', routerKey: 'x',
314
+ resolveAdapter: () => ({ ...boxAdapter, output: { parse: () => { throw new Error('parser failed') } } }),
315
+ resolveClient: () => fakeClient as never,
316
+ })
317
+ assert.equal(failed.perTask[0]?.measurement, 'unavailable')
318
+ assert.equal(failed.perTask[0]?.execution?.phase, 'started')
319
+ assert.deepEqual(failed.perTask[0]?.usage, {
320
+ input: 23, output: 7, costUsd: 0.04, tokensKnown: false, usdKnown: false,
321
+ }, `${failure} failure retains paid receipts without claiming complete accounting`)
322
+ assert.equal(failed.perTask[0]?.events?.length, failure === 'stream' ? 1 : 3, 'observed events are retained once')
323
+ assert.match(failed.perTask[0]?.detail ?? '', failure === 'stream' ? /stream disconnected/ : /parser failed/)
324
+ }
325
+ streamThrows = false
326
+ setupFails = true
327
+ const setupFailure = await runBenchmarks({
328
+ benchmarks: ['boxy'], cells: [{ label: 'setup-failure', model: 'm', backend: 'sandbox' }],
329
+ routerBaseUrl: 'x', routerKey: 'x', resolveAdapter: () => boxAdapter,
330
+ resolveClient: () => fakeClient as never,
331
+ })
332
+ assert.deepEqual(setupFailure.perTask[0]?.execution, { phase: 'unknown', terminalOutcome: 'unknown' })
333
+ assert.equal(setupFailure.perTask[0]?.measurement, 'unavailable')
334
+ assert.equal(setupFailure.perTask[0]?.usage?.tokensKnown, false, 'no observed receipt does not prove zero usage')
335
+ assert.equal(setupFailure.perTask[0]?.usage?.usdKnown, false)
336
+ assert.equal(setupFailure.perTask[0]?.events?.length, 0)
337
+ setupFails = false
338
+ for (const success of [false, true]) {
339
+ completed = success
340
+ text = ''
341
+ const empty = await runBenchmarks({
342
+ benchmarks: ['alpha'], cells: [{ label: 'empty', model: 'm', backend: 'sandbox' }],
343
+ routerBaseUrl: 'x', routerKey: 'x', n: 1, resolveAdapter: resolveStub,
344
+ resolveClient: () => fakeClient as never,
345
+ })
346
+ assert.equal(empty.perTask[0]?.ok, false)
347
+ assert.equal(empty.perTask[0]?.measurement, 'available', 'captured empty output is a measurable failure')
348
+ assert.equal(empty.perTask[0]?.execution?.terminalOutcome, success ? 'succeeded' : 'failed')
349
+ assert.equal(empty.rows[0]?.errored, 0)
350
+ assert.equal(empty.perTask[0]?.usage?.input, 23)
351
+ }
270
352
  }
271
353
 
272
354
  // An unavailable benchmark (preflight throws) is skipped, not fatal; the sweep still runs the rest.
@@ -310,4 +392,180 @@ async function main(): Promise<void> {
310
392
  console.log('run-benchmarks.test: OK (24-shot matrix, subset, reps, unavailable-skip, judge self-check, guards)')
311
393
  }
312
394
 
395
+ async function managedExecutionProof(): Promise<void> {
396
+ function fixture(options: { failure?: 'stream' | 'abort'; signal?: AbortSignal; onSecond?: () => void } = {}) {
397
+ const operations: string[] = []
398
+ const requests: Array<{ prompt: string; sessionId?: string }> = []
399
+ let creates = 0
400
+ let grades = 0
401
+ const client = {
402
+ async create() {
403
+ creates += 1
404
+ let patch = 'WRONG'
405
+ let turns = 0
406
+ return {
407
+ id: `managed-box-${creates}`,
408
+ async exec(command: string, opts?: { sessionId?: string }) {
409
+ operations.push(command)
410
+ assert.ok(opts?.sessionId, 'working checks and extraction address the worker session')
411
+ return { exitCode: 0, stdout: command === 'extract' ? patch : 'working check: repair the missing branch', stderr: '' }
412
+ },
413
+ async *streamPrompt(prompt: string, opts?: { sessionId?: string; signal?: AbortSignal }) {
414
+ turns += 1
415
+ requests.push({ prompt, sessionId: opts?.sessionId })
416
+ operations.push(`prompt:${turns}`)
417
+ // A repeated receipt id across prompts must not collapse paid calls.
418
+ yield { type: 'llm_call', data: { id: 'same-receipt-id', tokensIn: 11, tokensOut: 3, costUsd: 0.02 } }
419
+ if (turns === 2) {
420
+ options.onSecond?.()
421
+ if (options.failure === 'stream') throw new Error('second prompt disconnected')
422
+ if (options.failure === 'abort') {
423
+ assert.equal(opts?.signal?.aborted, true)
424
+ throw Object.assign(new Error('cancelled second prompt'), { name: 'AbortError' })
425
+ }
426
+ }
427
+ if (prompt.includes('repair the missing branch')) patch = 'PATCH'
428
+ yield { type: 'result', data: { finalText: patch, success: true, status: 'success' } }
429
+ yield { type: 'done', data: { outcome: { type: 'completed' } } }
430
+ },
431
+ async delete() { operations.push('delete') },
432
+ }
433
+ },
434
+ async criuStatus() { return { available: false } },
435
+ }
436
+ const adapter: BenchmarkAdapter = {
437
+ name: 'managed',
438
+ preflight: async () => {},
439
+ loadTasks: async () => [{ id: 'task', prompt: 'fix the branch', metadata: { gold: 'PRIVATE FINAL ORACLE' } }],
440
+ boxSetup: () => ({ command: 'setup' }),
441
+ boxExtract: () => ({ command: 'extract' }),
442
+ goldArtifact: async () => 'PATCH',
443
+ judge: async (_task, artifact) => {
444
+ grades += 1
445
+ operations.push('judge')
446
+ return { resolved: artifact === 'PATCH', score: artifact === 'PATCH' ? 1 : 0 }
447
+ },
448
+ }
449
+ return {
450
+ operations, requests, get creates() { return creates }, get grades() { return grades },
451
+ run: (execute?: BenchExecution, loopAttempts = 1) => runBenchmarks({
452
+ benchmarks: ['managed'], cells: [{ label: 'worker', model: 'model', backend: 'sandbox' }],
453
+ routerBaseUrl: 'unused', routerKey: 'unused', verifyJudge: false, loopAttempts,
454
+ ...(options.signal ? { signal: options.signal } : {}),
455
+ ...(execute ? { execute } : {}),
456
+ resolveAdapter: () => adapter, resolveClient: () => client as never,
457
+ }),
458
+ }
459
+ }
460
+ const correct: BenchExecution = async (context) => {
461
+ assert.deepEqual(Object.keys(context).sort(), ['attempt', 'benchmark', 'profile', 'prompt', 'run', 'signal', 'taskId'])
462
+ assert.equal('close' in context.run, false)
463
+ assert.equal(context.profile.model?.default, 'model')
464
+ const first = await context.run.start(context.prompt)
465
+ assert.equal(first.out, 'WRONG')
466
+ const feedback = await context.run.box.exec('working-check', { sessionId: context.run.sessionId })
467
+ await context.run.resume(feedback.stdout)
468
+ }
469
+ const enabled = fixture()
470
+ const report = await enabled.run(correct)
471
+ assert.equal(report.perTask[0]?.resolved, true)
472
+ assert.deepEqual(enabled.operations, ['setup', 'prompt:1', 'working-check', 'prompt:2', 'extract', 'delete', 'judge'])
473
+ assert.equal(enabled.creates, 1)
474
+ assert.equal(enabled.grades, 1)
475
+ assert.equal(enabled.requests[0]?.sessionId, enabled.requests[1]?.sessionId)
476
+ assert.equal(enabled.requests[1]?.prompt, 'working check: repair the missing branch')
477
+ assert.deepEqual(report.perTask[0]?.usage, { input: 22, output: 6, costUsd: 0.04 })
478
+ assert.deepEqual(report.perTask[0]?.prompts?.map((p) => [p.attempt, p.index, p.method, p.prompt]), [
479
+ [1, 0, 'start', 'fix the branch'], [1, 1, 'resume', 'working check: repair the missing branch'],
480
+ ])
481
+ assert.equal(report.perTask[0]?.events?.length, 6)
482
+ const disabled = fixture()
483
+ const withheld = await disabled.run(async ({ run, prompt }) => {
484
+ await run.start(prompt)
485
+ await run.resume('try again without a correction')
486
+ })
487
+ assert.equal(withheld.perTask[0]?.resolved, false, 'withholding the correction prevents the scripted repair')
488
+ assert.deepEqual(withheld.perTask[0]?.usage, report.perTask[0]?.usage, 'control spends the same scripted resources')
489
+ const defaultRun = await fixture().run()
490
+ assert.equal(defaultRun.perTask[0]?.prompts?.length, 1)
491
+ assert.deepEqual(defaultRun.perTask[0]?.usage, { input: 11, output: 3, costUsd: 0.02 })
492
+
493
+ for (const failure of ['stream', 'abort'] as const) {
494
+ const controller = new AbortController()
495
+ const broken = fixture({ failure, signal: controller.signal, onSecond: () => { if (failure === 'abort') controller.abort() } })
496
+ const failed = await broken.run(correct)
497
+ const row = failed.perTask[0]!
498
+ assert.equal(row.ok, false)
499
+ assert.equal(row.measurement, 'unavailable')
500
+ assert.equal(row.prompts?.length, 2)
501
+ assert.equal(row.events?.length, 4)
502
+ assert.deepEqual(row.usage, { input: 22, output: 6, costUsd: 0.04, tokensKnown: false, usdKnown: false })
503
+ assert.equal(row.prompts?.[0]?.usage.tokensKnown, undefined)
504
+ assert.equal(row.prompts?.[1]?.usage.tokensKnown, false)
505
+ assert.equal(broken.operations.filter((op) => op === 'delete').length, 1)
506
+ assert.equal(broken.operations.includes('extract'), false)
507
+ }
508
+ const policyFailure = fixture()
509
+ const policyFailed = await policyFailure.run(async ({ run, prompt }) => {
510
+ await run.start(prompt)
511
+ throw new Error('policy failed after paid work')
512
+ })
513
+ assert.equal(policyFailed.perTask[0]?.ok, false)
514
+ assert.deepEqual(policyFailed.perTask[0]?.usage, { input: 11, output: 3, costUsd: 0.02 })
515
+ assert.match(policyFailed.perTask[0]?.detail ?? '', /policy failed/)
516
+ assert.equal(policyFailure.operations.filter((op) => op === 'delete').length, 1)
517
+
518
+ const policyAbortController = new AbortController()
519
+ const policyAbort = fixture({ signal: policyAbortController.signal })
520
+ const abortedPolicy = await policyAbort.run(async ({ run, prompt }) => {
521
+ await run.start(prompt)
522
+ policyAbortController.abort()
523
+ await new Promise<void>(() => {})
524
+ })
525
+ assert.equal(abortedPolicy.perTask[0]?.ok, false, 'cancellation stops waiting for policy work')
526
+ assert.equal(abortedPolicy.perTask[0]?.prompts?.length, 1)
527
+ assert.deepEqual(abortedPolicy.perTask[0]?.usage, { input: 11, output: 3, costUsd: 0.02 })
528
+ assert.equal(policyAbort.operations.filter((op) => op === 'delete').length, 1)
529
+
530
+ const looped = fixture()
531
+ const retried = await looped.run(async ({ run, prompt, attempt }) => {
532
+ await run.start(prompt)
533
+ await run.resume(attempt === 2 ? 'repair the missing branch' : 'check again')
534
+ }, 2)
535
+ assert.equal(retried.perTask[0]?.resolved, true)
536
+ assert.equal(looped.creates, 2)
537
+ assert.deepEqual(retried.perTask[0]?.prompts?.map((p) => [p.attempt, p.index]), [[1, 0], [1, 1], [2, 0], [2, 1]])
538
+ assert.deepEqual(retried.perTask[0]?.usage, { input: 44, output: 12, costUsd: 0.08 })
539
+ assert.equal(looped.operations.filter((op) => op === 'setup').length, 2)
540
+ assert.equal(looped.operations.filter((op) => op === 'extract').length, 2)
541
+ assert.equal(looped.operations.filter((op) => op === 'delete').length, 2)
542
+
543
+ const unawaited = fixture()
544
+ const awaitedByOwner = await unawaited.run(async ({ run, prompt }) => { void run.start(prompt) })
545
+ assert.equal(awaitedByOwner.perTask[0]?.prompts?.length, 1)
546
+ assert.deepEqual(unawaited.operations, ['setup', 'prompt:1', 'extract', 'delete', 'judge'])
547
+ const overlapping = fixture()
548
+ const overlap = await overlapping.run(async ({ run, prompt }) => {
549
+ const first = run.start(prompt)
550
+ assert.throws(() => run.resume('overlap'), /sequential/)
551
+ await first
552
+ })
553
+ assert.equal(overlap.perTask[0]?.ok, false)
554
+ assert.equal(overlapping.operations.includes('extract'), false)
555
+ let retained: BenchExecutionContext['run'] | undefined
556
+ await fixture().run(async ({ run, prompt }) => { retained = run; await run.start(prompt) })
557
+ assert.throws(() => retained!.resume('too late'), /settled/)
558
+ const skipped = await fixture().run(async () => {})
559
+ assert.equal(skipped.perTask[0]?.ok, false)
560
+ assert.equal(skipped.perTask[0]?.prompts?.length, 0)
561
+ const passthrough: BenchExecution = async () => {}
562
+ let received: BenchExecution | undefined
563
+ await runBenchmarks({
564
+ benchmarks: ['alpha'], cells: [{ label: 'custom', model: 'm' }], n: 1,
565
+ routerBaseUrl: 'unused', routerKey: 'unused', resolveAdapter: resolveStub,
566
+ execute: passthrough, runShot: async ({ execute }) => { received = execute; return { artifact: '', ok: false } },
567
+ })
568
+ assert.equal(received, passthrough, 'custom shots choose how to consume the callback')
569
+ }
570
+
313
571
  void main()