@tangle-network/agent-bench 0.10.0 → 0.11.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-bench",
3
- "version": "0.10.0",
3
+ "version": "0.11.1",
4
4
  "type": "module",
5
5
  "description": "Benchmark adapters and execution for agent-runtime across coding, tool-use, RAG, memory, browser, and terminal tasks.",
6
6
  "repository": {
@@ -25,11 +25,11 @@
25
25
  }
26
26
  },
27
27
  "dependencies": {
28
- "@tangle-network/agent-eval": ">=0.178.0 <0.179.0",
28
+ "@tangle-network/agent-eval": ">=0.179.0 <0.180.0",
29
29
  "@tangle-network/agent-interface": "^2.6.0",
30
- "@tangle-network/agent-knowledge": "^15.0.1",
31
- "@tangle-network/sandbox": ">=0.36.4 <0.38.0",
32
- "@tangle-network/agent-runtime": "^0.203.0"
30
+ "@tangle-network/agent-knowledge": "^15.0.2",
31
+ "@tangle-network/sandbox": ">=0.36.4 <0.39.0",
32
+ "@tangle-network/agent-runtime": "^0.204.1"
33
33
  },
34
34
  "devDependencies": {
35
35
  "@arethetypeswrong/cli": "0.18.5",
@@ -165,6 +165,51 @@ writeFileSync(
165
165
  path.join(consumerDir, 'index.ts'),
166
166
  "import { createSweBenchAdapter, executePreparedPierCandidate, FilePierCandidateTrialController, resolveAdapter, runBenchmarks, runStagedJudge, StagedJudgeError, type BenchmarkAdapter, type JudgeArtifactReceipt, type PierCandidateTrialController, type PierCandidateTrialHandle, type PierDockerConnection, type StagedPierCandidateExecution } from '@tangle-network/agent-bench'\n\nconst adapter: BenchmarkAdapter = resolveAdapter('swe-bench')\nconst captureAdapter: BenchmarkAdapter = createSweBenchAdapter({ captureEvaluatorArtifacts: ({ taskId, attemptSequence }) => ({ destination: `/tmp/${taskId}/${attemptSequence}` }) })\nconst receipt = undefined as JudgeArtifactReceipt | undefined\nconst staged = undefined as StagedPierCandidateExecution | undefined\nconst trial = undefined as PierCandidateTrialHandle | undefined\nconst controller = undefined as PierCandidateTrialController | undefined\nconst dockerConnection = undefined as PierDockerConnection | undefined\nvoid adapter\nvoid captureAdapter\nvoid receipt\nvoid staged\nvoid trial\nvoid controller\nvoid dockerConnection\nvoid executePreparedPierCandidate\nvoid FilePierCandidateTrialController\nvoid runBenchmarks\nvoid runStagedJudge\nvoid StagedJudgeError\n",
167
167
  )
168
+ await writeFile(
169
+ path.join(consumerDir, 'managed-execution.ts'),
170
+ `import assert from 'node:assert/strict'
171
+ import { runBenchmarks, type BenchExecution, type BenchExecutionContext, type BenchPromptResult } from '@tangle-network/agent-bench'
172
+
173
+ const sessions: Array<string | undefined> = []
174
+ let closed = 0
175
+ const execute: BenchExecution = async (context: BenchExecutionContext) => {
176
+ await context.run.start(context.prompt)
177
+ await context.run.resume('corrected')
178
+ }
179
+ const report = await runBenchmarks({
180
+ benchmarks: ['fixture'], cells: [{ label: 'worker', model: 'fixture' }],
181
+ routerBaseUrl: 'unused', routerKey: 'unused', verifyJudge: false, execute,
182
+ resolveAdapter: () => ({
183
+ name: 'fixture', preflight: async () => {},
184
+ loadTasks: async () => [{ id: 'task', prompt: 'initial' }],
185
+ goldArtifact: async () => undefined,
186
+ judge: async (_task, artifact) => ({ resolved: artifact === 'corrected', score: artifact === 'corrected' ? 1 : 0 }),
187
+ }),
188
+ resolveClient: () => ({
189
+ criuStatus: async () => ({ available: false }),
190
+ create: async () => ({
191
+ id: 'packed-fixture',
192
+ async *streamPrompt(prompt: string, options?: { sessionId?: string }) {
193
+ sessions.push(options?.sessionId)
194
+ yield { type: 'llm_call', data: { tokensIn: 2, tokensOut: 1, costUsd: 0.01 } }
195
+ yield { type: 'result', data: { finalText: prompt, success: true, status: 'success' } }
196
+ yield { type: 'done', data: { outcome: { type: 'completed' } } }
197
+ },
198
+ delete: async () => { closed += 1 },
199
+ }),
200
+ }) as never,
201
+ })
202
+ const prompts: readonly BenchPromptResult[] = report.perTask[0]?.prompts ?? []
203
+ assert.equal(report.perTask[0]?.resolved, true)
204
+ assert.equal(prompts.length, 2)
205
+ assert.equal(prompts[1]?.method, 'resume')
206
+ assert.equal(prompts[1]?.prompt, 'corrected')
207
+ assert.equal(sessions[0], sessions[1])
208
+ assert.ok(sessions[0])
209
+ assert.equal(closed, 1)
210
+ assert.deepEqual(report.perTask[0]?.usage, { input: 4, output: 2, costUsd: 0.02 })
211
+ `,
212
+ )
168
213
  await writeFile(
169
214
  path.join(consumerDir, 'index.mjs'),
170
215
  "import { resolveAdapter, runBenchmarks } from '@tangle-network/agent-bench'\nimport { ADAPTERS } from '@tangle-network/agent-bench/adapters'\nimport { createCragAdapter } from '@tangle-network/agent-bench/benchmarks/crag'\n\nif (typeof resolveAdapter !== 'function' || typeof runBenchmarks !== 'function') throw new Error('root exports are not executable')\nif (typeof ADAPTERS !== 'object' || typeof createCragAdapter !== 'function') throw new Error('subpath exports are not executable')\nif (resolveAdapter('crag').name !== 'crag') throw new Error('compiled adapter registry returned the wrong adapter')\nprocess.env.TOOLLM_FIXTURES = '1'\nconst toolLlmTasks = await resolveAdapter('toollm').loadTasks({ limit: 1 })\nif (toolLlmTasks.length !== 1 || toolLlmTasks[0]?.id !== '1') throw new Error('packed ToolLLM fixture loading failed')\n",
@@ -189,7 +234,7 @@ if (!score.resolved || score.score !== 1) {
189
234
  `${JSON.stringify(
190
235
  {
191
236
  compilerOptions: publicTsconfig.compilerOptions,
192
- files: ['index.ts'],
237
+ files: ['index.ts', 'managed-execution.ts'],
193
238
  },
194
239
  null,
195
240
  2,
@@ -255,6 +300,7 @@ for name in sorted(expected):
255
300
  throw new Error(`expected TypeScript ${TYPESCRIPT_6}, received ${typescript6.stdout.trim()}`)
256
301
  }
257
302
  await run('npm', ['exec', '--', 'tsx', 'index.ts'], consumerDir)
303
+ await run('npm', ['exec', '--', 'tsx', 'managed-execution.ts'], consumerDir)
258
304
  const installedPackage = path.join(consumerDir, 'node_modules', '@tangle-network', 'agent-bench')
259
305
  const prepared = await run(
260
306
  'npm',
package/src/index.ts CHANGED
@@ -64,6 +64,9 @@ export {
64
64
  runBenchmarks,
65
65
  printBenchmarksReport,
66
66
  type BenchCell,
67
+ type BenchExecution,
68
+ type BenchExecutionContext,
69
+ type BenchPromptResult,
67
70
  type BenchShot,
68
71
  type BenchShotResult,
69
72
  type BenchCellTaskResult,
@@ -5,7 +5,7 @@
5
5
  */
6
6
  import assert from 'node:assert/strict'
7
7
  import type { BenchmarkAdapter, BenchScore, BenchTask } from './benchmarks/types'
8
- import { runBenchmarks, type BenchShot } from './run-benchmarks'
8
+ import { runBenchmarks, type BenchShot, type BenchExecution, type BenchExecutionContext } from './run-benchmarks'
9
9
 
10
10
  function stubAdapter(name: string, n: number): BenchmarkAdapter {
11
11
  const tasks: BenchTask[] = Array.from({ length: n }, (_, i) => ({
@@ -47,6 +47,7 @@ const shot: BenchShot = async ({ adapter, task, cell }) => {
47
47
  }
48
48
 
49
49
  async function main(): Promise<void> {
50
+ await managedExecutionProof()
50
51
  // Matrix: 2 benchmarks × 3 cells × 4 tasks = 24 shots.
51
52
  const report = await runBenchmarks({
52
53
  benchmarks: ['alpha', 'beta'],
@@ -391,4 +392,180 @@ async function main(): Promise<void> {
391
392
  console.log('run-benchmarks.test: OK (24-shot matrix, subset, reps, unavailable-skip, judge self-check, guards)')
392
393
  }
393
394
 
395
+ async function managedExecutionProof(): Promise<void> {
396
+ function fixture(options: { failure?: 'stream' | 'abort'; signal?: AbortSignal; onSecond?: () => void } = {}) {
397
+ const operations: string[] = []
398
+ const requests: Array<{ prompt: string; sessionId?: string }> = []
399
+ let creates = 0
400
+ let grades = 0
401
+ const client = {
402
+ async create() {
403
+ creates += 1
404
+ let patch = 'WRONG'
405
+ let turns = 0
406
+ return {
407
+ id: `managed-box-${creates}`,
408
+ async exec(command: string, opts?: { sessionId?: string }) {
409
+ operations.push(command)
410
+ assert.ok(opts?.sessionId, 'working checks and extraction address the worker session')
411
+ return { exitCode: 0, stdout: command === 'extract' ? patch : 'working check: repair the missing branch', stderr: '' }
412
+ },
413
+ async *streamPrompt(prompt: string, opts?: { sessionId?: string; signal?: AbortSignal }) {
414
+ turns += 1
415
+ requests.push({ prompt, sessionId: opts?.sessionId })
416
+ operations.push(`prompt:${turns}`)
417
+ // A repeated receipt id across prompts must not collapse paid calls.
418
+ yield { type: 'llm_call', data: { id: 'same-receipt-id', tokensIn: 11, tokensOut: 3, costUsd: 0.02 } }
419
+ if (turns === 2) {
420
+ options.onSecond?.()
421
+ if (options.failure === 'stream') throw new Error('second prompt disconnected')
422
+ if (options.failure === 'abort') {
423
+ assert.equal(opts?.signal?.aborted, true)
424
+ throw Object.assign(new Error('cancelled second prompt'), { name: 'AbortError' })
425
+ }
426
+ }
427
+ if (prompt.includes('repair the missing branch')) patch = 'PATCH'
428
+ yield { type: 'result', data: { finalText: patch, success: true, status: 'success' } }
429
+ yield { type: 'done', data: { outcome: { type: 'completed' } } }
430
+ },
431
+ async delete() { operations.push('delete') },
432
+ }
433
+ },
434
+ async criuStatus() { return { available: false } },
435
+ }
436
+ const adapter: BenchmarkAdapter = {
437
+ name: 'managed',
438
+ preflight: async () => {},
439
+ loadTasks: async () => [{ id: 'task', prompt: 'fix the branch', metadata: { gold: 'PRIVATE FINAL ORACLE' } }],
440
+ boxSetup: () => ({ command: 'setup' }),
441
+ boxExtract: () => ({ command: 'extract' }),
442
+ goldArtifact: async () => 'PATCH',
443
+ judge: async (_task, artifact) => {
444
+ grades += 1
445
+ operations.push('judge')
446
+ return { resolved: artifact === 'PATCH', score: artifact === 'PATCH' ? 1 : 0 }
447
+ },
448
+ }
449
+ return {
450
+ operations, requests, get creates() { return creates }, get grades() { return grades },
451
+ run: (execute?: BenchExecution, loopAttempts = 1) => runBenchmarks({
452
+ benchmarks: ['managed'], cells: [{ label: 'worker', model: 'model', backend: 'sandbox' }],
453
+ routerBaseUrl: 'unused', routerKey: 'unused', verifyJudge: false, loopAttempts,
454
+ ...(options.signal ? { signal: options.signal } : {}),
455
+ ...(execute ? { execute } : {}),
456
+ resolveAdapter: () => adapter, resolveClient: () => client as never,
457
+ }),
458
+ }
459
+ }
460
+ const correct: BenchExecution = async (context) => {
461
+ assert.deepEqual(Object.keys(context).sort(), ['attempt', 'benchmark', 'profile', 'prompt', 'run', 'signal', 'taskId'])
462
+ assert.equal('close' in context.run, false)
463
+ assert.equal(context.profile.model?.default, 'model')
464
+ const first = await context.run.start(context.prompt)
465
+ assert.equal(first.out, 'WRONG')
466
+ const feedback = await context.run.box.exec('working-check', { sessionId: context.run.sessionId })
467
+ await context.run.resume(feedback.stdout)
468
+ }
469
+ const enabled = fixture()
470
+ const report = await enabled.run(correct)
471
+ assert.equal(report.perTask[0]?.resolved, true)
472
+ assert.deepEqual(enabled.operations, ['setup', 'prompt:1', 'working-check', 'prompt:2', 'extract', 'delete', 'judge'])
473
+ assert.equal(enabled.creates, 1)
474
+ assert.equal(enabled.grades, 1)
475
+ assert.equal(enabled.requests[0]?.sessionId, enabled.requests[1]?.sessionId)
476
+ assert.equal(enabled.requests[1]?.prompt, 'working check: repair the missing branch')
477
+ assert.deepEqual(report.perTask[0]?.usage, { input: 22, output: 6, costUsd: 0.04 })
478
+ assert.deepEqual(report.perTask[0]?.prompts?.map((p) => [p.attempt, p.index, p.method, p.prompt]), [
479
+ [1, 0, 'start', 'fix the branch'], [1, 1, 'resume', 'working check: repair the missing branch'],
480
+ ])
481
+ assert.equal(report.perTask[0]?.events?.length, 6)
482
+ const disabled = fixture()
483
+ const withheld = await disabled.run(async ({ run, prompt }) => {
484
+ await run.start(prompt)
485
+ await run.resume('try again without a correction')
486
+ })
487
+ assert.equal(withheld.perTask[0]?.resolved, false, 'withholding the correction prevents the scripted repair')
488
+ assert.deepEqual(withheld.perTask[0]?.usage, report.perTask[0]?.usage, 'control spends the same scripted resources')
489
+ const defaultRun = await fixture().run()
490
+ assert.equal(defaultRun.perTask[0]?.prompts?.length, 1)
491
+ assert.deepEqual(defaultRun.perTask[0]?.usage, { input: 11, output: 3, costUsd: 0.02 })
492
+
493
+ for (const failure of ['stream', 'abort'] as const) {
494
+ const controller = new AbortController()
495
+ const broken = fixture({ failure, signal: controller.signal, onSecond: () => { if (failure === 'abort') controller.abort() } })
496
+ const failed = await broken.run(correct)
497
+ const row = failed.perTask[0]!
498
+ assert.equal(row.ok, false)
499
+ assert.equal(row.measurement, 'unavailable')
500
+ assert.equal(row.prompts?.length, 2)
501
+ assert.equal(row.events?.length, 4)
502
+ assert.deepEqual(row.usage, { input: 22, output: 6, costUsd: 0.04, tokensKnown: false, usdKnown: false })
503
+ assert.equal(row.prompts?.[0]?.usage.tokensKnown, undefined)
504
+ assert.equal(row.prompts?.[1]?.usage.tokensKnown, false)
505
+ assert.equal(broken.operations.filter((op) => op === 'delete').length, 1)
506
+ assert.equal(broken.operations.includes('extract'), false)
507
+ }
508
+ const policyFailure = fixture()
509
+ const policyFailed = await policyFailure.run(async ({ run, prompt }) => {
510
+ await run.start(prompt)
511
+ throw new Error('policy failed after paid work')
512
+ })
513
+ assert.equal(policyFailed.perTask[0]?.ok, false)
514
+ assert.deepEqual(policyFailed.perTask[0]?.usage, { input: 11, output: 3, costUsd: 0.02 })
515
+ assert.match(policyFailed.perTask[0]?.detail ?? '', /policy failed/)
516
+ assert.equal(policyFailure.operations.filter((op) => op === 'delete').length, 1)
517
+
518
+ const policyAbortController = new AbortController()
519
+ const policyAbort = fixture({ signal: policyAbortController.signal })
520
+ const abortedPolicy = await policyAbort.run(async ({ run, prompt }) => {
521
+ await run.start(prompt)
522
+ policyAbortController.abort()
523
+ await new Promise<void>(() => {})
524
+ })
525
+ assert.equal(abortedPolicy.perTask[0]?.ok, false, 'cancellation stops waiting for policy work')
526
+ assert.equal(abortedPolicy.perTask[0]?.prompts?.length, 1)
527
+ assert.deepEqual(abortedPolicy.perTask[0]?.usage, { input: 11, output: 3, costUsd: 0.02 })
528
+ assert.equal(policyAbort.operations.filter((op) => op === 'delete').length, 1)
529
+
530
+ const looped = fixture()
531
+ const retried = await looped.run(async ({ run, prompt, attempt }) => {
532
+ await run.start(prompt)
533
+ await run.resume(attempt === 2 ? 'repair the missing branch' : 'check again')
534
+ }, 2)
535
+ assert.equal(retried.perTask[0]?.resolved, true)
536
+ assert.equal(looped.creates, 2)
537
+ assert.deepEqual(retried.perTask[0]?.prompts?.map((p) => [p.attempt, p.index]), [[1, 0], [1, 1], [2, 0], [2, 1]])
538
+ assert.deepEqual(retried.perTask[0]?.usage, { input: 44, output: 12, costUsd: 0.08 })
539
+ assert.equal(looped.operations.filter((op) => op === 'setup').length, 2)
540
+ assert.equal(looped.operations.filter((op) => op === 'extract').length, 2)
541
+ assert.equal(looped.operations.filter((op) => op === 'delete').length, 2)
542
+
543
+ const unawaited = fixture()
544
+ const awaitedByOwner = await unawaited.run(async ({ run, prompt }) => { void run.start(prompt) })
545
+ assert.equal(awaitedByOwner.perTask[0]?.prompts?.length, 1)
546
+ assert.deepEqual(unawaited.operations, ['setup', 'prompt:1', 'extract', 'delete', 'judge'])
547
+ const overlapping = fixture()
548
+ const overlap = await overlapping.run(async ({ run, prompt }) => {
549
+ const first = run.start(prompt)
550
+ assert.throws(() => run.resume('overlap'), /sequential/)
551
+ await first
552
+ })
553
+ assert.equal(overlap.perTask[0]?.ok, false)
554
+ assert.equal(overlapping.operations.includes('extract'), false)
555
+ let retained: BenchExecutionContext['run'] | undefined
556
+ await fixture().run(async ({ run, prompt }) => { retained = run; await run.start(prompt) })
557
+ assert.throws(() => retained!.resume('too late'), /settled/)
558
+ const skipped = await fixture().run(async () => {})
559
+ assert.equal(skipped.perTask[0]?.ok, false)
560
+ assert.equal(skipped.perTask[0]?.prompts?.length, 0)
561
+ const passthrough: BenchExecution = async () => {}
562
+ let received: BenchExecution | undefined
563
+ await runBenchmarks({
564
+ benchmarks: ['alpha'], cells: [{ label: 'custom', model: 'm' }], n: 1,
565
+ routerBaseUrl: 'unused', routerKey: 'unused', resolveAdapter: resolveStub,
566
+ execute: passthrough, runShot: async ({ execute }) => { received = execute; return { artifact: '', ok: false } },
567
+ })
568
+ assert.equal(received, passthrough, 'custom shots choose how to consume the callback')
569
+ }
570
+
394
571
  void main()
@@ -35,6 +35,8 @@ import type {
35
35
  AgentRunSpec,
36
36
  Deliverable,
37
37
  OpenSandboxRunOptions,
38
+ SandboxRun,
39
+ TurnResult,
38
40
  } from '@tangle-network/agent-runtime/kernel'
39
41
  import { openSandboxRun, SandboxRunAbortError, sumSandboxUsage } from '@tangle-network/agent-runtime/kernel'
40
42
  import type { SandboxEvent } from '@tangle-network/sandbox'
@@ -61,6 +63,34 @@ export interface BenchCell {
61
63
  readonly profile?: AgentProfile
62
64
  }
63
65
 
66
+ /** Caller-owned work inside a managed benchmark shot. Final grading stays outside this callback. */
67
+ export interface BenchExecutionContext {
68
+ readonly prompt: string
69
+ readonly profile: AgentProfile
70
+ readonly benchmark: string
71
+ readonly taskId: string
72
+ readonly attempt: number
73
+ readonly signal: AbortSignal
74
+ /** Runtime owns session continuity. Bench owns capture, extraction, and close. */
75
+ readonly run: Pick<SandboxRun<string>, 'start' | 'resume' | 'box' | 'sessionId'>
76
+ }
77
+
78
+ export type BenchExecution = (context: BenchExecutionContext) => Promise<void>
79
+
80
+ /** One submitted sandbox prompt, including partial evidence when its invocation throws. */
81
+ export interface BenchPromptResult {
82
+ readonly attempt: number
83
+ readonly index: number
84
+ readonly method: 'start' | 'resume'
85
+ readonly prompt: string
86
+ readonly sessionId?: string
87
+ readonly outcome?: TurnResult<string>['outcome']
88
+ readonly events: readonly SandboxEvent[]
89
+ readonly usage: ReturnType<typeof sumSandboxUsage>
90
+ readonly readError?: string
91
+ readonly error?: string
92
+ }
93
+
64
94
  /** A worker's artifact and observed execution evidence, before external grading. */
65
95
  export interface BenchShotResult {
66
96
  readonly artifact: string
@@ -69,6 +99,7 @@ export interface BenchShotResult {
69
99
  /** Provider observations, including explicit unknown counters. Omitted when the shot reports none. */
70
100
  readonly usage?: ReturnType<typeof sumSandboxUsage>
71
101
  readonly events?: readonly SandboxEvent[]
102
+ readonly prompts?: readonly BenchPromptResult[]
72
103
  /** Observed dispatch and terminal state, independent of artifact quality. */
73
104
  readonly execution?: {
74
105
  readonly phase: 'not-started' | 'started' | 'unknown'
@@ -87,6 +118,8 @@ export type BenchShot = (input: {
87
118
  readonly prompt?: string
88
119
  /** 1-based attempt index for looped runs. */
89
120
  readonly attempt?: number
121
+ /** Custom shots own whether they consume this managed execution callback. */
122
+ readonly execute?: BenchExecution
90
123
  readonly routerBaseUrl: string
91
124
  readonly routerKey: string
92
125
  /** Optional inference credential for the box; routerKey continues to authorize sandbox control. */
@@ -131,6 +164,8 @@ export interface RunBenchmarksOptions {
131
164
  /** Self-verify each benchmark's judge against its gold artifact on the first task before spending
132
165
  * model tokens; a benchmark whose judge rejects its own gold is recorded unavailable. Default true. */
133
166
  readonly verifyJudge?: boolean
167
+ /** Caller policy inside each managed shot, including every refine-loop attempt. */
168
+ readonly execute?: BenchExecution
134
169
  /** Test seam: a deterministic shot runner. Defaults to the `openSandboxRun` leaf. */
135
170
  readonly runShot?: BenchShot
136
171
  /** Test seam: resolve a benchmark key to an adapter. Defaults to the registry `resolveAdapter`. */
@@ -157,6 +192,7 @@ export interface BenchCellTaskResult {
157
192
  readonly usage?: ReturnType<typeof sumSandboxUsage>
158
193
  /** Worker events only; benchmark grading remains outside this trace. */
159
194
  readonly events?: readonly SandboxEvent[]
195
+ readonly prompts?: readonly BenchPromptResult[]
160
196
  }
161
197
 
162
198
  export interface BenchLeaderboardRow {
@@ -204,7 +240,7 @@ function finalText(events: readonly SandboxEvent[]): string {
204
240
 
205
241
  /** The default real-agent shot: one `openSandboxRun` over the cell's harness+model, deliverable
206
242
  * extracted by the adapter's parser (or final text), abortable on `timeoutMs`. */
207
- const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerBaseUrl, routerKey, modelApiKey, bridgeUrl, bridgeBearer, sandboxBaseUrl, timeoutMs, signal, resolveClient }) => {
243
+ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, attempt = 1, execute, routerBaseUrl, routerKey, modelApiKey, bridgeUrl, bridgeBearer, sandboxBaseUrl, timeoutMs, signal, resolveClient }) => {
208
244
  signal?.throwIfAborted()
209
245
  const client = (resolveClient ?? resolveBenchClient)({
210
246
  backend: cell.backend ?? 'router',
@@ -272,12 +308,91 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
272
308
  }
273
309
  let run: Awaited<ReturnType<typeof openSandboxRun<string>>> | undefined
274
310
  let result: BenchShotResult = { artifact: '', ok: false }
311
+ const prompts: BenchPromptResult[] = []
312
+ const execution: {
313
+ lastTurn?: TurnResult<string>
314
+ active?: Promise<TurnResult<string>>
315
+ accepting: boolean
316
+ coordinationError?: Error
317
+ } = { accepting: true }
275
318
  try {
276
319
  run = await openSandboxRun(client, runOptions, deliverable)
277
320
  result = { ...result, execution: { phase: 'unknown', terminalOutcome: 'unknown' } }
278
- const turn = await run.start(prompt ?? task.prompt)
321
+ const managedRun = run
322
+ const invoke = (method: 'start' | 'resume', input: string): Promise<TurnResult<string>> => {
323
+ if (!execution.accepting || execution.active) {
324
+ execution.coordinationError = new Error(!execution.accepting
325
+ ? 'benchmark execution callback has settled'
326
+ : 'benchmark execution requires sequential start/resume calls')
327
+ throw execution.coordinationError
328
+ }
329
+ runOptions.signal.throwIfAborted()
330
+ const index = prompts.length
331
+ const eventStart = observedEvents.length
332
+ execution.lastTurn = undefined
333
+ const capture = async (): Promise<TurnResult<string>> => {
334
+ let turn: TurnResult<string> | undefined
335
+ let failure: unknown
336
+ try {
337
+ turn = await managedRun[method](input)
338
+ return turn
339
+ } catch (error) {
340
+ failure = error
341
+ throw error
342
+ } finally {
343
+ // The observer captures generic stream/parser failures as well as aborts.
344
+ const events = observedEvents.slice(eventStart)
345
+ if (turn) execution.lastTurn = { ...turn, events, outcome: { ...turn.outcome } }
346
+ let sessionId: string | undefined
347
+ try { sessionId = managedRun.sessionId } catch { /* Creation may have failed. */ }
348
+ const readError = turn?.readError ?? (failure instanceof SandboxRunAbortError ? failure.readError : undefined)
349
+ prompts.push({
350
+ attempt, index, method, prompt: input,
351
+ ...(sessionId === undefined ? {} : { sessionId }),
352
+ ...(turn === undefined ? {} : { outcome: { ...turn.outcome } }),
353
+ events,
354
+ usage: turn === undefined
355
+ ? { ...sumSandboxUsage(events), tokensKnown: false, usdKnown: false }
356
+ : sumSandboxUsage(events),
357
+ ...(readError === undefined ? {} : { readError }),
358
+ ...(turn !== undefined ? {} : { error: failure instanceof Error ? failure.message : String(failure) }),
359
+ })
360
+ }
361
+ }
362
+ const pending = capture()
363
+ execution.active = pending
364
+ // Observe even a caller's unawaited invocation; cleanup must await its capture.
365
+ void pending.then(() => { execution.active = undefined }, () => { execution.active = undefined })
366
+ return pending
367
+ }
368
+ const context: BenchExecutionContext = {
369
+ prompt: prompt ?? task.prompt, profile: structuredClone(profile), benchmark: adapter.name, taskId: task.id, attempt,
370
+ signal: runOptions.signal,
371
+ run: {
372
+ start: (input) => invoke('start', input),
373
+ resume: (input) => invoke('resume', input),
374
+ get box() { return managedRun.box },
375
+ get sessionId() { return managedRun.sessionId },
376
+ },
377
+ }
378
+ try {
379
+ await executeWithSignal(async () => {
380
+ if (execute) await execute(context)
381
+ else await context.run.start(context.prompt)
382
+ }, runOptions.signal)
383
+ } catch (error) {
384
+ if (execution.active) controller.abort()
385
+ throw error
386
+ } finally {
387
+ execution.accepting = false
388
+ await execution.active?.catch(() => undefined)
389
+ }
390
+ if (execution.coordinationError) throw execution.coordinationError
391
+ runOptions.signal.throwIfAborted()
392
+ const turn = execution.lastTurn
393
+ if (!turn) throw new Error(prompts.at(-1)?.error ?? 'benchmark execution returned without a completed prompt')
279
394
  result = {
280
- artifact: '', ok: false, usage: sumSandboxUsage(turn.events), events: turn.events,
395
+ artifact: '', ok: false, usage: promptUsage(prompts), events: observedEvents, prompts,
281
396
  execution: {
282
397
  phase: 'started',
283
398
  terminalOutcome: turn.outcome.success ? 'succeeded' : turn.outcome.status === 'failed' ? 'failed' : 'incomplete',
@@ -358,11 +473,12 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
358
473
  execution: result.execution,
359
474
  artifactAvailable: turn.readError === undefined && boxExtractError === undefined,
360
475
  usage: result.usage,
361
- events: turn.events,
476
+ events: observedEvents,
477
+ prompts,
362
478
  ...(detail ? { detail } : {}),
363
479
  }
364
480
  } catch (err) {
365
- const events = err instanceof SandboxRunAbortError ? err.events : observedEvents
481
+ const events = observedEvents
366
482
  result = {
367
483
  ...result,
368
484
  ok: false,
@@ -372,9 +488,9 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
372
488
  terminalOutcome: result.execution?.terminalOutcome ?? 'unknown',
373
489
  },
374
490
  detail: err instanceof Error ? err.message : String(err),
375
- // A thrown capture cannot establish that every paid receipt arrived.
376
- usage: { ...sumSandboxUsage(events), tokensKnown: false, usdKnown: false },
491
+ usage: promptUsage(prompts),
377
492
  events,
493
+ prompts,
378
494
  }
379
495
  } finally {
380
496
  if (timer) clearTimeout(timer)
@@ -388,6 +504,25 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
388
504
  return result
389
505
  }
390
506
 
507
+ /** Stop waiting on policy work when cancelled; the caller must also cancel its external effects. */
508
+ async function executeWithSignal(execute: () => Promise<void>, signal: AbortSignal): Promise<void> {
509
+ signal.throwIfAborted()
510
+ let onAbort: (() => void) | undefined
511
+ const aborted = new Promise<never>((_resolve, reject) => {
512
+ onAbort = () => reject(signal.reason ?? new Error('aborted'))
513
+ signal.addEventListener('abort', onAbort, { once: true })
514
+ })
515
+ try {
516
+ await Promise.race([Promise.resolve().then(execute), aborted])
517
+ } finally {
518
+ if (onAbort) signal.removeEventListener('abort', onAbort)
519
+ }
520
+ }
521
+
522
+ function promptUsage(prompts: readonly BenchPromptResult[]): ReturnType<typeof sumSandboxUsage> {
523
+ return combinedUsage(prompts.map((prompt) => ({ artifact: '', ok: false, usage: prompt.usage })))
524
+ }
525
+
391
526
  function parseMaybeJson(value: string): unknown {
392
527
  try {
393
528
  return JSON.parse(value) as unknown
@@ -484,6 +619,7 @@ async function loopedShot(
484
619
  artifactAvailable: false,
485
620
  usage: combinedUsage(pendingShot ? [...completed, { artifact: '', ok: false }] : completed),
486
621
  events: completed.flatMap((shot) => shot.events ?? []),
622
+ prompts: completed.flatMap((shot) => shot.prompts ?? []),
487
623
  detail: err instanceof Error ? err.message : String(err),
488
624
  }
489
625
  }
@@ -509,6 +645,7 @@ async function loopedShot(
509
645
  artifactAvailable: shots.get(best.round)?.artifactAvailable,
510
646
  usage: combinedUsage([...shots.values()]),
511
647
  events: [...shots.values()].flatMap((shot) => shot.events ?? []),
648
+ prompts: [...shots.values()].flatMap((shot) => shot.prompts ?? []),
512
649
  detail: JSON.stringify({
513
650
  mode: 'refine-loop',
514
651
  attempts: result.rounds.length,
@@ -639,6 +776,7 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
639
776
  ...(opts.timeoutMs ? { timeoutMs: opts.timeoutMs } : {}),
640
777
  ...(opts.signal ? { signal: opts.signal } : {}),
641
778
  ...(opts.resolveClient ? { resolveClient: opts.resolveClient } : {}),
779
+ ...(opts.execute ? { execute: opts.execute } : {}),
642
780
  }
643
781
  invoked = true
644
782
  out = loopAttempts > 1 ? await loopedShot(shotInput, shot, loopAttempts) : await shot(shotInput)
@@ -658,6 +796,7 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
658
796
  artifact: out.artifact,
659
797
  ...(out.usage === undefined ? {} : { usage: out.usage }),
660
798
  ...(out.events === undefined ? {} : { events: out.events }),
799
+ ...(out.prompts === undefined ? {} : { prompts: out.prompts }),
661
800
  }
662
801
  } catch (err) {
663
802
  // Missing results do not prove that dispatch or paid inference never occurred.
@@ -679,6 +818,7 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
679
818
  ...(out === undefined ? {} : { artifact: out.artifact }),
680
819
  ...(out?.usage === undefined ? {} : { usage: out.usage }),
681
820
  ...(out?.events === undefined ? {} : { events: out.events }),
821
+ ...(out?.prompts === undefined ? {} : { prompts: out.prompts }),
682
822
  }
683
823
  }
684
824
  void index