@tangle-network/agent-bench 0.10.0 → 0.11.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +10 -0
- package/HARNESS.md +43 -5
- package/README.md +22 -0
- package/dist/index.d.ts +33 -2
- package/dist/index.js +133 -15
- package/dist/index.js.map +1 -1
- package/package.json +5 -5
- package/scripts/verify-packed-consumer.mjs +47 -1
- package/src/index.ts +3 -0
- package/src/run-benchmarks.test.mts +178 -1
- package/src/run-benchmarks.ts +147 -7
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-bench",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.11.1",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Benchmark adapters and execution for agent-runtime across coding, tool-use, RAG, memory, browser, and terminal tasks.",
|
|
6
6
|
"repository": {
|
|
@@ -25,11 +25,11 @@
|
|
|
25
25
|
}
|
|
26
26
|
},
|
|
27
27
|
"dependencies": {
|
|
28
|
-
"@tangle-network/agent-eval": ">=0.
|
|
28
|
+
"@tangle-network/agent-eval": ">=0.179.0 <0.180.0",
|
|
29
29
|
"@tangle-network/agent-interface": "^2.6.0",
|
|
30
|
-
"@tangle-network/agent-knowledge": "^15.0.
|
|
31
|
-
"@tangle-network/sandbox": ">=0.36.4 <0.
|
|
32
|
-
"@tangle-network/agent-runtime": "^0.
|
|
30
|
+
"@tangle-network/agent-knowledge": "^15.0.2",
|
|
31
|
+
"@tangle-network/sandbox": ">=0.36.4 <0.39.0",
|
|
32
|
+
"@tangle-network/agent-runtime": "^0.204.1"
|
|
33
33
|
},
|
|
34
34
|
"devDependencies": {
|
|
35
35
|
"@arethetypeswrong/cli": "0.18.5",
|
|
@@ -165,6 +165,51 @@ writeFileSync(
|
|
|
165
165
|
path.join(consumerDir, 'index.ts'),
|
|
166
166
|
"import { createSweBenchAdapter, executePreparedPierCandidate, FilePierCandidateTrialController, resolveAdapter, runBenchmarks, runStagedJudge, StagedJudgeError, type BenchmarkAdapter, type JudgeArtifactReceipt, type PierCandidateTrialController, type PierCandidateTrialHandle, type PierDockerConnection, type StagedPierCandidateExecution } from '@tangle-network/agent-bench'\n\nconst adapter: BenchmarkAdapter = resolveAdapter('swe-bench')\nconst captureAdapter: BenchmarkAdapter = createSweBenchAdapter({ captureEvaluatorArtifacts: ({ taskId, attemptSequence }) => ({ destination: `/tmp/${taskId}/${attemptSequence}` }) })\nconst receipt = undefined as JudgeArtifactReceipt | undefined\nconst staged = undefined as StagedPierCandidateExecution | undefined\nconst trial = undefined as PierCandidateTrialHandle | undefined\nconst controller = undefined as PierCandidateTrialController | undefined\nconst dockerConnection = undefined as PierDockerConnection | undefined\nvoid adapter\nvoid captureAdapter\nvoid receipt\nvoid staged\nvoid trial\nvoid controller\nvoid dockerConnection\nvoid executePreparedPierCandidate\nvoid FilePierCandidateTrialController\nvoid runBenchmarks\nvoid runStagedJudge\nvoid StagedJudgeError\n",
|
|
167
167
|
)
|
|
168
|
+
await writeFile(
|
|
169
|
+
path.join(consumerDir, 'managed-execution.ts'),
|
|
170
|
+
`import assert from 'node:assert/strict'
|
|
171
|
+
import { runBenchmarks, type BenchExecution, type BenchExecutionContext, type BenchPromptResult } from '@tangle-network/agent-bench'
|
|
172
|
+
|
|
173
|
+
const sessions: Array<string | undefined> = []
|
|
174
|
+
let closed = 0
|
|
175
|
+
const execute: BenchExecution = async (context: BenchExecutionContext) => {
|
|
176
|
+
await context.run.start(context.prompt)
|
|
177
|
+
await context.run.resume('corrected')
|
|
178
|
+
}
|
|
179
|
+
const report = await runBenchmarks({
|
|
180
|
+
benchmarks: ['fixture'], cells: [{ label: 'worker', model: 'fixture' }],
|
|
181
|
+
routerBaseUrl: 'unused', routerKey: 'unused', verifyJudge: false, execute,
|
|
182
|
+
resolveAdapter: () => ({
|
|
183
|
+
name: 'fixture', preflight: async () => {},
|
|
184
|
+
loadTasks: async () => [{ id: 'task', prompt: 'initial' }],
|
|
185
|
+
goldArtifact: async () => undefined,
|
|
186
|
+
judge: async (_task, artifact) => ({ resolved: artifact === 'corrected', score: artifact === 'corrected' ? 1 : 0 }),
|
|
187
|
+
}),
|
|
188
|
+
resolveClient: () => ({
|
|
189
|
+
criuStatus: async () => ({ available: false }),
|
|
190
|
+
create: async () => ({
|
|
191
|
+
id: 'packed-fixture',
|
|
192
|
+
async *streamPrompt(prompt: string, options?: { sessionId?: string }) {
|
|
193
|
+
sessions.push(options?.sessionId)
|
|
194
|
+
yield { type: 'llm_call', data: { tokensIn: 2, tokensOut: 1, costUsd: 0.01 } }
|
|
195
|
+
yield { type: 'result', data: { finalText: prompt, success: true, status: 'success' } }
|
|
196
|
+
yield { type: 'done', data: { outcome: { type: 'completed' } } }
|
|
197
|
+
},
|
|
198
|
+
delete: async () => { closed += 1 },
|
|
199
|
+
}),
|
|
200
|
+
}) as never,
|
|
201
|
+
})
|
|
202
|
+
const prompts: readonly BenchPromptResult[] = report.perTask[0]?.prompts ?? []
|
|
203
|
+
assert.equal(report.perTask[0]?.resolved, true)
|
|
204
|
+
assert.equal(prompts.length, 2)
|
|
205
|
+
assert.equal(prompts[1]?.method, 'resume')
|
|
206
|
+
assert.equal(prompts[1]?.prompt, 'corrected')
|
|
207
|
+
assert.equal(sessions[0], sessions[1])
|
|
208
|
+
assert.ok(sessions[0])
|
|
209
|
+
assert.equal(closed, 1)
|
|
210
|
+
assert.deepEqual(report.perTask[0]?.usage, { input: 4, output: 2, costUsd: 0.02 })
|
|
211
|
+
`,
|
|
212
|
+
)
|
|
168
213
|
await writeFile(
|
|
169
214
|
path.join(consumerDir, 'index.mjs'),
|
|
170
215
|
"import { resolveAdapter, runBenchmarks } from '@tangle-network/agent-bench'\nimport { ADAPTERS } from '@tangle-network/agent-bench/adapters'\nimport { createCragAdapter } from '@tangle-network/agent-bench/benchmarks/crag'\n\nif (typeof resolveAdapter !== 'function' || typeof runBenchmarks !== 'function') throw new Error('root exports are not executable')\nif (typeof ADAPTERS !== 'object' || typeof createCragAdapter !== 'function') throw new Error('subpath exports are not executable')\nif (resolveAdapter('crag').name !== 'crag') throw new Error('compiled adapter registry returned the wrong adapter')\nprocess.env.TOOLLM_FIXTURES = '1'\nconst toolLlmTasks = await resolveAdapter('toollm').loadTasks({ limit: 1 })\nif (toolLlmTasks.length !== 1 || toolLlmTasks[0]?.id !== '1') throw new Error('packed ToolLLM fixture loading failed')\n",
|
|
@@ -189,7 +234,7 @@ if (!score.resolved || score.score !== 1) {
|
|
|
189
234
|
`${JSON.stringify(
|
|
190
235
|
{
|
|
191
236
|
compilerOptions: publicTsconfig.compilerOptions,
|
|
192
|
-
files: ['index.ts'],
|
|
237
|
+
files: ['index.ts', 'managed-execution.ts'],
|
|
193
238
|
},
|
|
194
239
|
null,
|
|
195
240
|
2,
|
|
@@ -255,6 +300,7 @@ for name in sorted(expected):
|
|
|
255
300
|
throw new Error(`expected TypeScript ${TYPESCRIPT_6}, received ${typescript6.stdout.trim()}`)
|
|
256
301
|
}
|
|
257
302
|
await run('npm', ['exec', '--', 'tsx', 'index.ts'], consumerDir)
|
|
303
|
+
await run('npm', ['exec', '--', 'tsx', 'managed-execution.ts'], consumerDir)
|
|
258
304
|
const installedPackage = path.join(consumerDir, 'node_modules', '@tangle-network', 'agent-bench')
|
|
259
305
|
const prepared = await run(
|
|
260
306
|
'npm',
|
package/src/index.ts
CHANGED
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
*/
|
|
6
6
|
import assert from 'node:assert/strict'
|
|
7
7
|
import type { BenchmarkAdapter, BenchScore, BenchTask } from './benchmarks/types'
|
|
8
|
-
import { runBenchmarks, type BenchShot } from './run-benchmarks'
|
|
8
|
+
import { runBenchmarks, type BenchShot, type BenchExecution, type BenchExecutionContext } from './run-benchmarks'
|
|
9
9
|
|
|
10
10
|
function stubAdapter(name: string, n: number): BenchmarkAdapter {
|
|
11
11
|
const tasks: BenchTask[] = Array.from({ length: n }, (_, i) => ({
|
|
@@ -47,6 +47,7 @@ const shot: BenchShot = async ({ adapter, task, cell }) => {
|
|
|
47
47
|
}
|
|
48
48
|
|
|
49
49
|
async function main(): Promise<void> {
|
|
50
|
+
await managedExecutionProof()
|
|
50
51
|
// Matrix: 2 benchmarks × 3 cells × 4 tasks = 24 shots.
|
|
51
52
|
const report = await runBenchmarks({
|
|
52
53
|
benchmarks: ['alpha', 'beta'],
|
|
@@ -391,4 +392,180 @@ async function main(): Promise<void> {
|
|
|
391
392
|
console.log('run-benchmarks.test: OK (24-shot matrix, subset, reps, unavailable-skip, judge self-check, guards)')
|
|
392
393
|
}
|
|
393
394
|
|
|
395
|
+
async function managedExecutionProof(): Promise<void> {
|
|
396
|
+
function fixture(options: { failure?: 'stream' | 'abort'; signal?: AbortSignal; onSecond?: () => void } = {}) {
|
|
397
|
+
const operations: string[] = []
|
|
398
|
+
const requests: Array<{ prompt: string; sessionId?: string }> = []
|
|
399
|
+
let creates = 0
|
|
400
|
+
let grades = 0
|
|
401
|
+
const client = {
|
|
402
|
+
async create() {
|
|
403
|
+
creates += 1
|
|
404
|
+
let patch = 'WRONG'
|
|
405
|
+
let turns = 0
|
|
406
|
+
return {
|
|
407
|
+
id: `managed-box-${creates}`,
|
|
408
|
+
async exec(command: string, opts?: { sessionId?: string }) {
|
|
409
|
+
operations.push(command)
|
|
410
|
+
assert.ok(opts?.sessionId, 'working checks and extraction address the worker session')
|
|
411
|
+
return { exitCode: 0, stdout: command === 'extract' ? patch : 'working check: repair the missing branch', stderr: '' }
|
|
412
|
+
},
|
|
413
|
+
async *streamPrompt(prompt: string, opts?: { sessionId?: string; signal?: AbortSignal }) {
|
|
414
|
+
turns += 1
|
|
415
|
+
requests.push({ prompt, sessionId: opts?.sessionId })
|
|
416
|
+
operations.push(`prompt:${turns}`)
|
|
417
|
+
// A repeated receipt id across prompts must not collapse paid calls.
|
|
418
|
+
yield { type: 'llm_call', data: { id: 'same-receipt-id', tokensIn: 11, tokensOut: 3, costUsd: 0.02 } }
|
|
419
|
+
if (turns === 2) {
|
|
420
|
+
options.onSecond?.()
|
|
421
|
+
if (options.failure === 'stream') throw new Error('second prompt disconnected')
|
|
422
|
+
if (options.failure === 'abort') {
|
|
423
|
+
assert.equal(opts?.signal?.aborted, true)
|
|
424
|
+
throw Object.assign(new Error('cancelled second prompt'), { name: 'AbortError' })
|
|
425
|
+
}
|
|
426
|
+
}
|
|
427
|
+
if (prompt.includes('repair the missing branch')) patch = 'PATCH'
|
|
428
|
+
yield { type: 'result', data: { finalText: patch, success: true, status: 'success' } }
|
|
429
|
+
yield { type: 'done', data: { outcome: { type: 'completed' } } }
|
|
430
|
+
},
|
|
431
|
+
async delete() { operations.push('delete') },
|
|
432
|
+
}
|
|
433
|
+
},
|
|
434
|
+
async criuStatus() { return { available: false } },
|
|
435
|
+
}
|
|
436
|
+
const adapter: BenchmarkAdapter = {
|
|
437
|
+
name: 'managed',
|
|
438
|
+
preflight: async () => {},
|
|
439
|
+
loadTasks: async () => [{ id: 'task', prompt: 'fix the branch', metadata: { gold: 'PRIVATE FINAL ORACLE' } }],
|
|
440
|
+
boxSetup: () => ({ command: 'setup' }),
|
|
441
|
+
boxExtract: () => ({ command: 'extract' }),
|
|
442
|
+
goldArtifact: async () => 'PATCH',
|
|
443
|
+
judge: async (_task, artifact) => {
|
|
444
|
+
grades += 1
|
|
445
|
+
operations.push('judge')
|
|
446
|
+
return { resolved: artifact === 'PATCH', score: artifact === 'PATCH' ? 1 : 0 }
|
|
447
|
+
},
|
|
448
|
+
}
|
|
449
|
+
return {
|
|
450
|
+
operations, requests, get creates() { return creates }, get grades() { return grades },
|
|
451
|
+
run: (execute?: BenchExecution, loopAttempts = 1) => runBenchmarks({
|
|
452
|
+
benchmarks: ['managed'], cells: [{ label: 'worker', model: 'model', backend: 'sandbox' }],
|
|
453
|
+
routerBaseUrl: 'unused', routerKey: 'unused', verifyJudge: false, loopAttempts,
|
|
454
|
+
...(options.signal ? { signal: options.signal } : {}),
|
|
455
|
+
...(execute ? { execute } : {}),
|
|
456
|
+
resolveAdapter: () => adapter, resolveClient: () => client as never,
|
|
457
|
+
}),
|
|
458
|
+
}
|
|
459
|
+
}
|
|
460
|
+
const correct: BenchExecution = async (context) => {
|
|
461
|
+
assert.deepEqual(Object.keys(context).sort(), ['attempt', 'benchmark', 'profile', 'prompt', 'run', 'signal', 'taskId'])
|
|
462
|
+
assert.equal('close' in context.run, false)
|
|
463
|
+
assert.equal(context.profile.model?.default, 'model')
|
|
464
|
+
const first = await context.run.start(context.prompt)
|
|
465
|
+
assert.equal(first.out, 'WRONG')
|
|
466
|
+
const feedback = await context.run.box.exec('working-check', { sessionId: context.run.sessionId })
|
|
467
|
+
await context.run.resume(feedback.stdout)
|
|
468
|
+
}
|
|
469
|
+
const enabled = fixture()
|
|
470
|
+
const report = await enabled.run(correct)
|
|
471
|
+
assert.equal(report.perTask[0]?.resolved, true)
|
|
472
|
+
assert.deepEqual(enabled.operations, ['setup', 'prompt:1', 'working-check', 'prompt:2', 'extract', 'delete', 'judge'])
|
|
473
|
+
assert.equal(enabled.creates, 1)
|
|
474
|
+
assert.equal(enabled.grades, 1)
|
|
475
|
+
assert.equal(enabled.requests[0]?.sessionId, enabled.requests[1]?.sessionId)
|
|
476
|
+
assert.equal(enabled.requests[1]?.prompt, 'working check: repair the missing branch')
|
|
477
|
+
assert.deepEqual(report.perTask[0]?.usage, { input: 22, output: 6, costUsd: 0.04 })
|
|
478
|
+
assert.deepEqual(report.perTask[0]?.prompts?.map((p) => [p.attempt, p.index, p.method, p.prompt]), [
|
|
479
|
+
[1, 0, 'start', 'fix the branch'], [1, 1, 'resume', 'working check: repair the missing branch'],
|
|
480
|
+
])
|
|
481
|
+
assert.equal(report.perTask[0]?.events?.length, 6)
|
|
482
|
+
const disabled = fixture()
|
|
483
|
+
const withheld = await disabled.run(async ({ run, prompt }) => {
|
|
484
|
+
await run.start(prompt)
|
|
485
|
+
await run.resume('try again without a correction')
|
|
486
|
+
})
|
|
487
|
+
assert.equal(withheld.perTask[0]?.resolved, false, 'withholding the correction prevents the scripted repair')
|
|
488
|
+
assert.deepEqual(withheld.perTask[0]?.usage, report.perTask[0]?.usage, 'control spends the same scripted resources')
|
|
489
|
+
const defaultRun = await fixture().run()
|
|
490
|
+
assert.equal(defaultRun.perTask[0]?.prompts?.length, 1)
|
|
491
|
+
assert.deepEqual(defaultRun.perTask[0]?.usage, { input: 11, output: 3, costUsd: 0.02 })
|
|
492
|
+
|
|
493
|
+
for (const failure of ['stream', 'abort'] as const) {
|
|
494
|
+
const controller = new AbortController()
|
|
495
|
+
const broken = fixture({ failure, signal: controller.signal, onSecond: () => { if (failure === 'abort') controller.abort() } })
|
|
496
|
+
const failed = await broken.run(correct)
|
|
497
|
+
const row = failed.perTask[0]!
|
|
498
|
+
assert.equal(row.ok, false)
|
|
499
|
+
assert.equal(row.measurement, 'unavailable')
|
|
500
|
+
assert.equal(row.prompts?.length, 2)
|
|
501
|
+
assert.equal(row.events?.length, 4)
|
|
502
|
+
assert.deepEqual(row.usage, { input: 22, output: 6, costUsd: 0.04, tokensKnown: false, usdKnown: false })
|
|
503
|
+
assert.equal(row.prompts?.[0]?.usage.tokensKnown, undefined)
|
|
504
|
+
assert.equal(row.prompts?.[1]?.usage.tokensKnown, false)
|
|
505
|
+
assert.equal(broken.operations.filter((op) => op === 'delete').length, 1)
|
|
506
|
+
assert.equal(broken.operations.includes('extract'), false)
|
|
507
|
+
}
|
|
508
|
+
const policyFailure = fixture()
|
|
509
|
+
const policyFailed = await policyFailure.run(async ({ run, prompt }) => {
|
|
510
|
+
await run.start(prompt)
|
|
511
|
+
throw new Error('policy failed after paid work')
|
|
512
|
+
})
|
|
513
|
+
assert.equal(policyFailed.perTask[0]?.ok, false)
|
|
514
|
+
assert.deepEqual(policyFailed.perTask[0]?.usage, { input: 11, output: 3, costUsd: 0.02 })
|
|
515
|
+
assert.match(policyFailed.perTask[0]?.detail ?? '', /policy failed/)
|
|
516
|
+
assert.equal(policyFailure.operations.filter((op) => op === 'delete').length, 1)
|
|
517
|
+
|
|
518
|
+
const policyAbortController = new AbortController()
|
|
519
|
+
const policyAbort = fixture({ signal: policyAbortController.signal })
|
|
520
|
+
const abortedPolicy = await policyAbort.run(async ({ run, prompt }) => {
|
|
521
|
+
await run.start(prompt)
|
|
522
|
+
policyAbortController.abort()
|
|
523
|
+
await new Promise<void>(() => {})
|
|
524
|
+
})
|
|
525
|
+
assert.equal(abortedPolicy.perTask[0]?.ok, false, 'cancellation stops waiting for policy work')
|
|
526
|
+
assert.equal(abortedPolicy.perTask[0]?.prompts?.length, 1)
|
|
527
|
+
assert.deepEqual(abortedPolicy.perTask[0]?.usage, { input: 11, output: 3, costUsd: 0.02 })
|
|
528
|
+
assert.equal(policyAbort.operations.filter((op) => op === 'delete').length, 1)
|
|
529
|
+
|
|
530
|
+
const looped = fixture()
|
|
531
|
+
const retried = await looped.run(async ({ run, prompt, attempt }) => {
|
|
532
|
+
await run.start(prompt)
|
|
533
|
+
await run.resume(attempt === 2 ? 'repair the missing branch' : 'check again')
|
|
534
|
+
}, 2)
|
|
535
|
+
assert.equal(retried.perTask[0]?.resolved, true)
|
|
536
|
+
assert.equal(looped.creates, 2)
|
|
537
|
+
assert.deepEqual(retried.perTask[0]?.prompts?.map((p) => [p.attempt, p.index]), [[1, 0], [1, 1], [2, 0], [2, 1]])
|
|
538
|
+
assert.deepEqual(retried.perTask[0]?.usage, { input: 44, output: 12, costUsd: 0.08 })
|
|
539
|
+
assert.equal(looped.operations.filter((op) => op === 'setup').length, 2)
|
|
540
|
+
assert.equal(looped.operations.filter((op) => op === 'extract').length, 2)
|
|
541
|
+
assert.equal(looped.operations.filter((op) => op === 'delete').length, 2)
|
|
542
|
+
|
|
543
|
+
const unawaited = fixture()
|
|
544
|
+
const awaitedByOwner = await unawaited.run(async ({ run, prompt }) => { void run.start(prompt) })
|
|
545
|
+
assert.equal(awaitedByOwner.perTask[0]?.prompts?.length, 1)
|
|
546
|
+
assert.deepEqual(unawaited.operations, ['setup', 'prompt:1', 'extract', 'delete', 'judge'])
|
|
547
|
+
const overlapping = fixture()
|
|
548
|
+
const overlap = await overlapping.run(async ({ run, prompt }) => {
|
|
549
|
+
const first = run.start(prompt)
|
|
550
|
+
assert.throws(() => run.resume('overlap'), /sequential/)
|
|
551
|
+
await first
|
|
552
|
+
})
|
|
553
|
+
assert.equal(overlap.perTask[0]?.ok, false)
|
|
554
|
+
assert.equal(overlapping.operations.includes('extract'), false)
|
|
555
|
+
let retained: BenchExecutionContext['run'] | undefined
|
|
556
|
+
await fixture().run(async ({ run, prompt }) => { retained = run; await run.start(prompt) })
|
|
557
|
+
assert.throws(() => retained!.resume('too late'), /settled/)
|
|
558
|
+
const skipped = await fixture().run(async () => {})
|
|
559
|
+
assert.equal(skipped.perTask[0]?.ok, false)
|
|
560
|
+
assert.equal(skipped.perTask[0]?.prompts?.length, 0)
|
|
561
|
+
const passthrough: BenchExecution = async () => {}
|
|
562
|
+
let received: BenchExecution | undefined
|
|
563
|
+
await runBenchmarks({
|
|
564
|
+
benchmarks: ['alpha'], cells: [{ label: 'custom', model: 'm' }], n: 1,
|
|
565
|
+
routerBaseUrl: 'unused', routerKey: 'unused', resolveAdapter: resolveStub,
|
|
566
|
+
execute: passthrough, runShot: async ({ execute }) => { received = execute; return { artifact: '', ok: false } },
|
|
567
|
+
})
|
|
568
|
+
assert.equal(received, passthrough, 'custom shots choose how to consume the callback')
|
|
569
|
+
}
|
|
570
|
+
|
|
394
571
|
void main()
|
package/src/run-benchmarks.ts
CHANGED
|
@@ -35,6 +35,8 @@ import type {
|
|
|
35
35
|
AgentRunSpec,
|
|
36
36
|
Deliverable,
|
|
37
37
|
OpenSandboxRunOptions,
|
|
38
|
+
SandboxRun,
|
|
39
|
+
TurnResult,
|
|
38
40
|
} from '@tangle-network/agent-runtime/kernel'
|
|
39
41
|
import { openSandboxRun, SandboxRunAbortError, sumSandboxUsage } from '@tangle-network/agent-runtime/kernel'
|
|
40
42
|
import type { SandboxEvent } from '@tangle-network/sandbox'
|
|
@@ -61,6 +63,34 @@ export interface BenchCell {
|
|
|
61
63
|
readonly profile?: AgentProfile
|
|
62
64
|
}
|
|
63
65
|
|
|
66
|
+
/** Caller-owned work inside a managed benchmark shot. Final grading stays outside this callback. */
|
|
67
|
+
export interface BenchExecutionContext {
|
|
68
|
+
readonly prompt: string
|
|
69
|
+
readonly profile: AgentProfile
|
|
70
|
+
readonly benchmark: string
|
|
71
|
+
readonly taskId: string
|
|
72
|
+
readonly attempt: number
|
|
73
|
+
readonly signal: AbortSignal
|
|
74
|
+
/** Runtime owns session continuity. Bench owns capture, extraction, and close. */
|
|
75
|
+
readonly run: Pick<SandboxRun<string>, 'start' | 'resume' | 'box' | 'sessionId'>
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
export type BenchExecution = (context: BenchExecutionContext) => Promise<void>
|
|
79
|
+
|
|
80
|
+
/** One submitted sandbox prompt, including partial evidence when its invocation throws. */
|
|
81
|
+
export interface BenchPromptResult {
|
|
82
|
+
readonly attempt: number
|
|
83
|
+
readonly index: number
|
|
84
|
+
readonly method: 'start' | 'resume'
|
|
85
|
+
readonly prompt: string
|
|
86
|
+
readonly sessionId?: string
|
|
87
|
+
readonly outcome?: TurnResult<string>['outcome']
|
|
88
|
+
readonly events: readonly SandboxEvent[]
|
|
89
|
+
readonly usage: ReturnType<typeof sumSandboxUsage>
|
|
90
|
+
readonly readError?: string
|
|
91
|
+
readonly error?: string
|
|
92
|
+
}
|
|
93
|
+
|
|
64
94
|
/** A worker's artifact and observed execution evidence, before external grading. */
|
|
65
95
|
export interface BenchShotResult {
|
|
66
96
|
readonly artifact: string
|
|
@@ -69,6 +99,7 @@ export interface BenchShotResult {
|
|
|
69
99
|
/** Provider observations, including explicit unknown counters. Omitted when the shot reports none. */
|
|
70
100
|
readonly usage?: ReturnType<typeof sumSandboxUsage>
|
|
71
101
|
readonly events?: readonly SandboxEvent[]
|
|
102
|
+
readonly prompts?: readonly BenchPromptResult[]
|
|
72
103
|
/** Observed dispatch and terminal state, independent of artifact quality. */
|
|
73
104
|
readonly execution?: {
|
|
74
105
|
readonly phase: 'not-started' | 'started' | 'unknown'
|
|
@@ -87,6 +118,8 @@ export type BenchShot = (input: {
|
|
|
87
118
|
readonly prompt?: string
|
|
88
119
|
/** 1-based attempt index for looped runs. */
|
|
89
120
|
readonly attempt?: number
|
|
121
|
+
/** Custom shots own whether they consume this managed execution callback. */
|
|
122
|
+
readonly execute?: BenchExecution
|
|
90
123
|
readonly routerBaseUrl: string
|
|
91
124
|
readonly routerKey: string
|
|
92
125
|
/** Optional inference credential for the box; routerKey continues to authorize sandbox control. */
|
|
@@ -131,6 +164,8 @@ export interface RunBenchmarksOptions {
|
|
|
131
164
|
/** Self-verify each benchmark's judge against its gold artifact on the first task before spending
|
|
132
165
|
* model tokens; a benchmark whose judge rejects its own gold is recorded unavailable. Default true. */
|
|
133
166
|
readonly verifyJudge?: boolean
|
|
167
|
+
/** Caller policy inside each managed shot, including every refine-loop attempt. */
|
|
168
|
+
readonly execute?: BenchExecution
|
|
134
169
|
/** Test seam: a deterministic shot runner. Defaults to the `openSandboxRun` leaf. */
|
|
135
170
|
readonly runShot?: BenchShot
|
|
136
171
|
/** Test seam: resolve a benchmark key to an adapter. Defaults to the registry `resolveAdapter`. */
|
|
@@ -157,6 +192,7 @@ export interface BenchCellTaskResult {
|
|
|
157
192
|
readonly usage?: ReturnType<typeof sumSandboxUsage>
|
|
158
193
|
/** Worker events only; benchmark grading remains outside this trace. */
|
|
159
194
|
readonly events?: readonly SandboxEvent[]
|
|
195
|
+
readonly prompts?: readonly BenchPromptResult[]
|
|
160
196
|
}
|
|
161
197
|
|
|
162
198
|
export interface BenchLeaderboardRow {
|
|
@@ -204,7 +240,7 @@ function finalText(events: readonly SandboxEvent[]): string {
|
|
|
204
240
|
|
|
205
241
|
/** The default real-agent shot: one `openSandboxRun` over the cell's harness+model, deliverable
|
|
206
242
|
* extracted by the adapter's parser (or final text), abortable on `timeoutMs`. */
|
|
207
|
-
const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerBaseUrl, routerKey, modelApiKey, bridgeUrl, bridgeBearer, sandboxBaseUrl, timeoutMs, signal, resolveClient }) => {
|
|
243
|
+
const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, attempt = 1, execute, routerBaseUrl, routerKey, modelApiKey, bridgeUrl, bridgeBearer, sandboxBaseUrl, timeoutMs, signal, resolveClient }) => {
|
|
208
244
|
signal?.throwIfAborted()
|
|
209
245
|
const client = (resolveClient ?? resolveBenchClient)({
|
|
210
246
|
backend: cell.backend ?? 'router',
|
|
@@ -272,12 +308,91 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
|
|
|
272
308
|
}
|
|
273
309
|
let run: Awaited<ReturnType<typeof openSandboxRun<string>>> | undefined
|
|
274
310
|
let result: BenchShotResult = { artifact: '', ok: false }
|
|
311
|
+
const prompts: BenchPromptResult[] = []
|
|
312
|
+
const execution: {
|
|
313
|
+
lastTurn?: TurnResult<string>
|
|
314
|
+
active?: Promise<TurnResult<string>>
|
|
315
|
+
accepting: boolean
|
|
316
|
+
coordinationError?: Error
|
|
317
|
+
} = { accepting: true }
|
|
275
318
|
try {
|
|
276
319
|
run = await openSandboxRun(client, runOptions, deliverable)
|
|
277
320
|
result = { ...result, execution: { phase: 'unknown', terminalOutcome: 'unknown' } }
|
|
278
|
-
const
|
|
321
|
+
const managedRun = run
|
|
322
|
+
const invoke = (method: 'start' | 'resume', input: string): Promise<TurnResult<string>> => {
|
|
323
|
+
if (!execution.accepting || execution.active) {
|
|
324
|
+
execution.coordinationError = new Error(!execution.accepting
|
|
325
|
+
? 'benchmark execution callback has settled'
|
|
326
|
+
: 'benchmark execution requires sequential start/resume calls')
|
|
327
|
+
throw execution.coordinationError
|
|
328
|
+
}
|
|
329
|
+
runOptions.signal.throwIfAborted()
|
|
330
|
+
const index = prompts.length
|
|
331
|
+
const eventStart = observedEvents.length
|
|
332
|
+
execution.lastTurn = undefined
|
|
333
|
+
const capture = async (): Promise<TurnResult<string>> => {
|
|
334
|
+
let turn: TurnResult<string> | undefined
|
|
335
|
+
let failure: unknown
|
|
336
|
+
try {
|
|
337
|
+
turn = await managedRun[method](input)
|
|
338
|
+
return turn
|
|
339
|
+
} catch (error) {
|
|
340
|
+
failure = error
|
|
341
|
+
throw error
|
|
342
|
+
} finally {
|
|
343
|
+
// The observer captures generic stream/parser failures as well as aborts.
|
|
344
|
+
const events = observedEvents.slice(eventStart)
|
|
345
|
+
if (turn) execution.lastTurn = { ...turn, events, outcome: { ...turn.outcome } }
|
|
346
|
+
let sessionId: string | undefined
|
|
347
|
+
try { sessionId = managedRun.sessionId } catch { /* Creation may have failed. */ }
|
|
348
|
+
const readError = turn?.readError ?? (failure instanceof SandboxRunAbortError ? failure.readError : undefined)
|
|
349
|
+
prompts.push({
|
|
350
|
+
attempt, index, method, prompt: input,
|
|
351
|
+
...(sessionId === undefined ? {} : { sessionId }),
|
|
352
|
+
...(turn === undefined ? {} : { outcome: { ...turn.outcome } }),
|
|
353
|
+
events,
|
|
354
|
+
usage: turn === undefined
|
|
355
|
+
? { ...sumSandboxUsage(events), tokensKnown: false, usdKnown: false }
|
|
356
|
+
: sumSandboxUsage(events),
|
|
357
|
+
...(readError === undefined ? {} : { readError }),
|
|
358
|
+
...(turn !== undefined ? {} : { error: failure instanceof Error ? failure.message : String(failure) }),
|
|
359
|
+
})
|
|
360
|
+
}
|
|
361
|
+
}
|
|
362
|
+
const pending = capture()
|
|
363
|
+
execution.active = pending
|
|
364
|
+
// Observe even a caller's unawaited invocation; cleanup must await its capture.
|
|
365
|
+
void pending.then(() => { execution.active = undefined }, () => { execution.active = undefined })
|
|
366
|
+
return pending
|
|
367
|
+
}
|
|
368
|
+
const context: BenchExecutionContext = {
|
|
369
|
+
prompt: prompt ?? task.prompt, profile: structuredClone(profile), benchmark: adapter.name, taskId: task.id, attempt,
|
|
370
|
+
signal: runOptions.signal,
|
|
371
|
+
run: {
|
|
372
|
+
start: (input) => invoke('start', input),
|
|
373
|
+
resume: (input) => invoke('resume', input),
|
|
374
|
+
get box() { return managedRun.box },
|
|
375
|
+
get sessionId() { return managedRun.sessionId },
|
|
376
|
+
},
|
|
377
|
+
}
|
|
378
|
+
try {
|
|
379
|
+
await executeWithSignal(async () => {
|
|
380
|
+
if (execute) await execute(context)
|
|
381
|
+
else await context.run.start(context.prompt)
|
|
382
|
+
}, runOptions.signal)
|
|
383
|
+
} catch (error) {
|
|
384
|
+
if (execution.active) controller.abort()
|
|
385
|
+
throw error
|
|
386
|
+
} finally {
|
|
387
|
+
execution.accepting = false
|
|
388
|
+
await execution.active?.catch(() => undefined)
|
|
389
|
+
}
|
|
390
|
+
if (execution.coordinationError) throw execution.coordinationError
|
|
391
|
+
runOptions.signal.throwIfAborted()
|
|
392
|
+
const turn = execution.lastTurn
|
|
393
|
+
if (!turn) throw new Error(prompts.at(-1)?.error ?? 'benchmark execution returned without a completed prompt')
|
|
279
394
|
result = {
|
|
280
|
-
artifact: '', ok: false, usage:
|
|
395
|
+
artifact: '', ok: false, usage: promptUsage(prompts), events: observedEvents, prompts,
|
|
281
396
|
execution: {
|
|
282
397
|
phase: 'started',
|
|
283
398
|
terminalOutcome: turn.outcome.success ? 'succeeded' : turn.outcome.status === 'failed' ? 'failed' : 'incomplete',
|
|
@@ -358,11 +473,12 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
|
|
|
358
473
|
execution: result.execution,
|
|
359
474
|
artifactAvailable: turn.readError === undefined && boxExtractError === undefined,
|
|
360
475
|
usage: result.usage,
|
|
361
|
-
events:
|
|
476
|
+
events: observedEvents,
|
|
477
|
+
prompts,
|
|
362
478
|
...(detail ? { detail } : {}),
|
|
363
479
|
}
|
|
364
480
|
} catch (err) {
|
|
365
|
-
const events =
|
|
481
|
+
const events = observedEvents
|
|
366
482
|
result = {
|
|
367
483
|
...result,
|
|
368
484
|
ok: false,
|
|
@@ -372,9 +488,9 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
|
|
|
372
488
|
terminalOutcome: result.execution?.terminalOutcome ?? 'unknown',
|
|
373
489
|
},
|
|
374
490
|
detail: err instanceof Error ? err.message : String(err),
|
|
375
|
-
|
|
376
|
-
usage: { ...sumSandboxUsage(events), tokensKnown: false, usdKnown: false },
|
|
491
|
+
usage: promptUsage(prompts),
|
|
377
492
|
events,
|
|
493
|
+
prompts,
|
|
378
494
|
}
|
|
379
495
|
} finally {
|
|
380
496
|
if (timer) clearTimeout(timer)
|
|
@@ -388,6 +504,25 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
|
|
|
388
504
|
return result
|
|
389
505
|
}
|
|
390
506
|
|
|
507
|
+
/** Stop waiting on policy work when cancelled; the caller must also cancel its external effects. */
|
|
508
|
+
async function executeWithSignal(execute: () => Promise<void>, signal: AbortSignal): Promise<void> {
|
|
509
|
+
signal.throwIfAborted()
|
|
510
|
+
let onAbort: (() => void) | undefined
|
|
511
|
+
const aborted = new Promise<never>((_resolve, reject) => {
|
|
512
|
+
onAbort = () => reject(signal.reason ?? new Error('aborted'))
|
|
513
|
+
signal.addEventListener('abort', onAbort, { once: true })
|
|
514
|
+
})
|
|
515
|
+
try {
|
|
516
|
+
await Promise.race([Promise.resolve().then(execute), aborted])
|
|
517
|
+
} finally {
|
|
518
|
+
if (onAbort) signal.removeEventListener('abort', onAbort)
|
|
519
|
+
}
|
|
520
|
+
}
|
|
521
|
+
|
|
522
|
+
function promptUsage(prompts: readonly BenchPromptResult[]): ReturnType<typeof sumSandboxUsage> {
|
|
523
|
+
return combinedUsage(prompts.map((prompt) => ({ artifact: '', ok: false, usage: prompt.usage })))
|
|
524
|
+
}
|
|
525
|
+
|
|
391
526
|
function parseMaybeJson(value: string): unknown {
|
|
392
527
|
try {
|
|
393
528
|
return JSON.parse(value) as unknown
|
|
@@ -484,6 +619,7 @@ async function loopedShot(
|
|
|
484
619
|
artifactAvailable: false,
|
|
485
620
|
usage: combinedUsage(pendingShot ? [...completed, { artifact: '', ok: false }] : completed),
|
|
486
621
|
events: completed.flatMap((shot) => shot.events ?? []),
|
|
622
|
+
prompts: completed.flatMap((shot) => shot.prompts ?? []),
|
|
487
623
|
detail: err instanceof Error ? err.message : String(err),
|
|
488
624
|
}
|
|
489
625
|
}
|
|
@@ -509,6 +645,7 @@ async function loopedShot(
|
|
|
509
645
|
artifactAvailable: shots.get(best.round)?.artifactAvailable,
|
|
510
646
|
usage: combinedUsage([...shots.values()]),
|
|
511
647
|
events: [...shots.values()].flatMap((shot) => shot.events ?? []),
|
|
648
|
+
prompts: [...shots.values()].flatMap((shot) => shot.prompts ?? []),
|
|
512
649
|
detail: JSON.stringify({
|
|
513
650
|
mode: 'refine-loop',
|
|
514
651
|
attempts: result.rounds.length,
|
|
@@ -639,6 +776,7 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
|
|
|
639
776
|
...(opts.timeoutMs ? { timeoutMs: opts.timeoutMs } : {}),
|
|
640
777
|
...(opts.signal ? { signal: opts.signal } : {}),
|
|
641
778
|
...(opts.resolveClient ? { resolveClient: opts.resolveClient } : {}),
|
|
779
|
+
...(opts.execute ? { execute: opts.execute } : {}),
|
|
642
780
|
}
|
|
643
781
|
invoked = true
|
|
644
782
|
out = loopAttempts > 1 ? await loopedShot(shotInput, shot, loopAttempts) : await shot(shotInput)
|
|
@@ -658,6 +796,7 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
|
|
|
658
796
|
artifact: out.artifact,
|
|
659
797
|
...(out.usage === undefined ? {} : { usage: out.usage }),
|
|
660
798
|
...(out.events === undefined ? {} : { events: out.events }),
|
|
799
|
+
...(out.prompts === undefined ? {} : { prompts: out.prompts }),
|
|
661
800
|
}
|
|
662
801
|
} catch (err) {
|
|
663
802
|
// Missing results do not prove that dispatch or paid inference never occurred.
|
|
@@ -679,6 +818,7 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
|
|
|
679
818
|
...(out === undefined ? {} : { artifact: out.artifact }),
|
|
680
819
|
...(out?.usage === undefined ? {} : { usage: out.usage }),
|
|
681
820
|
...(out?.events === undefined ? {} : { events: out.events }),
|
|
821
|
+
...(out?.prompts === undefined ? {} : { prompts: out.prompts }),
|
|
682
822
|
}
|
|
683
823
|
}
|
|
684
824
|
void index
|