@tangle-network/agent-bench 0.9.4 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +6 -0
- package/HARNESS.md +53 -15
- package/README.md +22 -0
- package/dist/index.d.ts +45 -5
- package/dist/index.js +180 -16
- package/dist/index.js.map +1 -1
- package/package.json +5 -5
- package/scripts/run-package-tests.mjs +12 -2
- package/scripts/run-package-tests.test.mjs +26 -1
- package/scripts/verify-packed-consumer.mjs +47 -1
- package/src/index.ts +3 -0
- package/src/run-benchmarks.test.mts +262 -4
- package/src/run-benchmarks.ts +195 -12
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-bench",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.11.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Benchmark adapters and execution for agent-runtime across coding, tool-use, RAG, memory, browser, and terminal tasks.",
|
|
6
6
|
"repository": {
|
|
@@ -25,11 +25,11 @@
|
|
|
25
25
|
}
|
|
26
26
|
},
|
|
27
27
|
"dependencies": {
|
|
28
|
-
"@tangle-network/agent-eval": ">=0.
|
|
29
|
-
"@tangle-network/agent-interface": "^2.
|
|
30
|
-
"@tangle-network/agent-knowledge": "^15.0.
|
|
28
|
+
"@tangle-network/agent-eval": ">=0.178.0 <0.179.0",
|
|
29
|
+
"@tangle-network/agent-interface": "^2.6.0",
|
|
30
|
+
"@tangle-network/agent-knowledge": "^15.0.1",
|
|
31
31
|
"@tangle-network/sandbox": ">=0.36.4 <0.38.0",
|
|
32
|
-
"@tangle-network/agent-runtime": "^0.
|
|
32
|
+
"@tangle-network/agent-runtime": "^0.203.2"
|
|
33
33
|
},
|
|
34
34
|
"devDependencies": {
|
|
35
35
|
"@arethetypeswrong/cli": "0.18.5",
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { execFile } from 'node:child_process'
|
|
2
|
-
import { access, readdir, readFile } from 'node:fs/promises'
|
|
2
|
+
import { access, readdir, readFile, realpath } from 'node:fs/promises'
|
|
3
|
+
import { tmpdir } from 'node:os'
|
|
3
4
|
import path from 'node:path'
|
|
4
5
|
import { fileURLToPath } from 'node:url'
|
|
5
6
|
import { promisify } from 'node:util'
|
|
@@ -64,6 +65,15 @@ export async function run(command, args, env = process.env) {
|
|
|
64
65
|
}
|
|
65
66
|
}
|
|
66
67
|
|
|
68
|
+
export async function runPythonTests(python, env = process.env) {
|
|
69
|
+
// Fixture roots must be physical paths; the production boundary rejects symlinked ancestors.
|
|
70
|
+
const physicalTemp = await realpath(env.TMPDIR || tmpdir())
|
|
71
|
+
await run(python, ['-m', 'unittest', 'discover', '-s', 'pier_agents', '-p', '*_test.py'], {
|
|
72
|
+
...env,
|
|
73
|
+
TMPDIR: physicalTemp,
|
|
74
|
+
})
|
|
75
|
+
}
|
|
76
|
+
|
|
67
77
|
async function main() {
|
|
68
78
|
const python = path.join(benchDir, '.venv', 'bin', 'python')
|
|
69
79
|
try {
|
|
@@ -101,7 +111,7 @@ async function main() {
|
|
|
101
111
|
await run('npx', ['vitest', 'run', ...vitestTests.map((file) => path.relative(benchDir, file))])
|
|
102
112
|
}
|
|
103
113
|
|
|
104
|
-
await
|
|
114
|
+
await runPythonTests(python)
|
|
105
115
|
|
|
106
116
|
console.log(
|
|
107
117
|
`package tests passed: ${tests.length}/${tests.length} TypeScript files (${nodeTests.length} node:test + ${vitestTests.length} vitest) + Pier bridge`,
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
import assert from 'node:assert/strict'
|
|
2
2
|
import { test } from 'node:test'
|
|
3
|
-
import {
|
|
3
|
+
import { mkdtemp, mkdir, readFile, realpath, rm, symlink, writeFile } from 'node:fs/promises'
|
|
4
|
+
import { tmpdir } from 'node:os'
|
|
5
|
+
import path from 'node:path'
|
|
6
|
+
import { packageNodeTestArgs, resolvePackageTestTimeoutMs, run, runPythonTests } from './run-package-tests.mjs'
|
|
4
7
|
|
|
5
8
|
test('package test concurrency reaches Node without changing selected files', () => {
|
|
6
9
|
const files = ['src/first.test.mts', 'src/second.test.ts']
|
|
@@ -45,3 +48,25 @@ test('package test timeout reaches the child process', async () => {
|
|
|
45
48
|
)
|
|
46
49
|
assert.ok(Date.now() - startedAt < 2_000)
|
|
47
50
|
})
|
|
51
|
+
|
|
52
|
+
test('Python fixtures use physical temporary paths without relaxing symlink guards', async () => {
|
|
53
|
+
const root = await mkdtemp(path.join(await realpath(tmpdir()), 'bench-python-temp-'))
|
|
54
|
+
try {
|
|
55
|
+
const target = path.join(root, 'physical')
|
|
56
|
+
const linked = path.join(root, 'linked')
|
|
57
|
+
const recorded = path.join(root, 'recorded.json')
|
|
58
|
+
const launcher = path.join(root, 'python-fixture')
|
|
59
|
+
await mkdir(target)
|
|
60
|
+
await symlink(target, linked, 'dir')
|
|
61
|
+
await writeFile(launcher, `#!${process.execPath}
|
|
62
|
+
const fs = require('node:fs'); fs.writeFileSync(process.env.RECORDED, JSON.stringify({ temporary: process.env.TMPDIR, args: process.argv.slice(2) }))
|
|
63
|
+
`, { mode: 0o755 })
|
|
64
|
+
await runPythonTests(launcher, { ...process.env, TMPDIR: linked, RECORDED: recorded })
|
|
65
|
+
assert.deepEqual(JSON.parse(await readFile(recorded, 'utf8')), {
|
|
66
|
+
temporary: target,
|
|
67
|
+
args: ['-m', 'unittest', 'discover', '-s', 'pier_agents', '-p', '*_test.py'],
|
|
68
|
+
})
|
|
69
|
+
} finally {
|
|
70
|
+
await rm(root, { recursive: true, force: true })
|
|
71
|
+
}
|
|
72
|
+
})
|
|
@@ -165,6 +165,51 @@ writeFileSync(
|
|
|
165
165
|
path.join(consumerDir, 'index.ts'),
|
|
166
166
|
"import { createSweBenchAdapter, executePreparedPierCandidate, FilePierCandidateTrialController, resolveAdapter, runBenchmarks, runStagedJudge, StagedJudgeError, type BenchmarkAdapter, type JudgeArtifactReceipt, type PierCandidateTrialController, type PierCandidateTrialHandle, type PierDockerConnection, type StagedPierCandidateExecution } from '@tangle-network/agent-bench'\n\nconst adapter: BenchmarkAdapter = resolveAdapter('swe-bench')\nconst captureAdapter: BenchmarkAdapter = createSweBenchAdapter({ captureEvaluatorArtifacts: ({ taskId, attemptSequence }) => ({ destination: `/tmp/${taskId}/${attemptSequence}` }) })\nconst receipt = undefined as JudgeArtifactReceipt | undefined\nconst staged = undefined as StagedPierCandidateExecution | undefined\nconst trial = undefined as PierCandidateTrialHandle | undefined\nconst controller = undefined as PierCandidateTrialController | undefined\nconst dockerConnection = undefined as PierDockerConnection | undefined\nvoid adapter\nvoid captureAdapter\nvoid receipt\nvoid staged\nvoid trial\nvoid controller\nvoid dockerConnection\nvoid executePreparedPierCandidate\nvoid FilePierCandidateTrialController\nvoid runBenchmarks\nvoid runStagedJudge\nvoid StagedJudgeError\n",
|
|
167
167
|
)
|
|
168
|
+
await writeFile(
|
|
169
|
+
path.join(consumerDir, 'managed-execution.ts'),
|
|
170
|
+
`import assert from 'node:assert/strict'
|
|
171
|
+
import { runBenchmarks, type BenchExecution, type BenchExecutionContext, type BenchPromptResult } from '@tangle-network/agent-bench'
|
|
172
|
+
|
|
173
|
+
const sessions: Array<string | undefined> = []
|
|
174
|
+
let closed = 0
|
|
175
|
+
const execute: BenchExecution = async (context: BenchExecutionContext) => {
|
|
176
|
+
await context.run.start(context.prompt)
|
|
177
|
+
await context.run.resume('corrected')
|
|
178
|
+
}
|
|
179
|
+
const report = await runBenchmarks({
|
|
180
|
+
benchmarks: ['fixture'], cells: [{ label: 'worker', model: 'fixture' }],
|
|
181
|
+
routerBaseUrl: 'unused', routerKey: 'unused', verifyJudge: false, execute,
|
|
182
|
+
resolveAdapter: () => ({
|
|
183
|
+
name: 'fixture', preflight: async () => {},
|
|
184
|
+
loadTasks: async () => [{ id: 'task', prompt: 'initial' }],
|
|
185
|
+
goldArtifact: async () => undefined,
|
|
186
|
+
judge: async (_task, artifact) => ({ resolved: artifact === 'corrected', score: artifact === 'corrected' ? 1 : 0 }),
|
|
187
|
+
}),
|
|
188
|
+
resolveClient: () => ({
|
|
189
|
+
criuStatus: async () => ({ available: false }),
|
|
190
|
+
create: async () => ({
|
|
191
|
+
id: 'packed-fixture',
|
|
192
|
+
async *streamPrompt(prompt: string, options?: { sessionId?: string }) {
|
|
193
|
+
sessions.push(options?.sessionId)
|
|
194
|
+
yield { type: 'llm_call', data: { tokensIn: 2, tokensOut: 1, costUsd: 0.01 } }
|
|
195
|
+
yield { type: 'result', data: { finalText: prompt, success: true, status: 'success' } }
|
|
196
|
+
yield { type: 'done', data: { outcome: { type: 'completed' } } }
|
|
197
|
+
},
|
|
198
|
+
delete: async () => { closed += 1 },
|
|
199
|
+
}),
|
|
200
|
+
}) as never,
|
|
201
|
+
})
|
|
202
|
+
const prompts: readonly BenchPromptResult[] = report.perTask[0]?.prompts ?? []
|
|
203
|
+
assert.equal(report.perTask[0]?.resolved, true)
|
|
204
|
+
assert.equal(prompts.length, 2)
|
|
205
|
+
assert.equal(prompts[1]?.method, 'resume')
|
|
206
|
+
assert.equal(prompts[1]?.prompt, 'corrected')
|
|
207
|
+
assert.equal(sessions[0], sessions[1])
|
|
208
|
+
assert.ok(sessions[0])
|
|
209
|
+
assert.equal(closed, 1)
|
|
210
|
+
assert.deepEqual(report.perTask[0]?.usage, { input: 4, output: 2, costUsd: 0.02 })
|
|
211
|
+
`,
|
|
212
|
+
)
|
|
168
213
|
await writeFile(
|
|
169
214
|
path.join(consumerDir, 'index.mjs'),
|
|
170
215
|
"import { resolveAdapter, runBenchmarks } from '@tangle-network/agent-bench'\nimport { ADAPTERS } from '@tangle-network/agent-bench/adapters'\nimport { createCragAdapter } from '@tangle-network/agent-bench/benchmarks/crag'\n\nif (typeof resolveAdapter !== 'function' || typeof runBenchmarks !== 'function') throw new Error('root exports are not executable')\nif (typeof ADAPTERS !== 'object' || typeof createCragAdapter !== 'function') throw new Error('subpath exports are not executable')\nif (resolveAdapter('crag').name !== 'crag') throw new Error('compiled adapter registry returned the wrong adapter')\nprocess.env.TOOLLM_FIXTURES = '1'\nconst toolLlmTasks = await resolveAdapter('toollm').loadTasks({ limit: 1 })\nif (toolLlmTasks.length !== 1 || toolLlmTasks[0]?.id !== '1') throw new Error('packed ToolLLM fixture loading failed')\n",
|
|
@@ -189,7 +234,7 @@ if (!score.resolved || score.score !== 1) {
|
|
|
189
234
|
`${JSON.stringify(
|
|
190
235
|
{
|
|
191
236
|
compilerOptions: publicTsconfig.compilerOptions,
|
|
192
|
-
files: ['index.ts'],
|
|
237
|
+
files: ['index.ts', 'managed-execution.ts'],
|
|
193
238
|
},
|
|
194
239
|
null,
|
|
195
240
|
2,
|
|
@@ -255,6 +300,7 @@ for name in sorted(expected):
|
|
|
255
300
|
throw new Error(`expected TypeScript ${TYPESCRIPT_6}, received ${typescript6.stdout.trim()}`)
|
|
256
301
|
}
|
|
257
302
|
await run('npm', ['exec', '--', 'tsx', 'index.ts'], consumerDir)
|
|
303
|
+
await run('npm', ['exec', '--', 'tsx', 'managed-execution.ts'], consumerDir)
|
|
258
304
|
const installedPackage = path.join(consumerDir, 'node_modules', '@tangle-network', 'agent-bench')
|
|
259
305
|
const prepared = await run(
|
|
260
306
|
'npm',
|
package/src/index.ts
CHANGED
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
*/
|
|
6
6
|
import assert from 'node:assert/strict'
|
|
7
7
|
import type { BenchmarkAdapter, BenchScore, BenchTask } from './benchmarks/types'
|
|
8
|
-
import { runBenchmarks, type BenchShot } from './run-benchmarks'
|
|
8
|
+
import { runBenchmarks, type BenchShot, type BenchExecution, type BenchExecutionContext } from './run-benchmarks'
|
|
9
9
|
|
|
10
10
|
function stubAdapter(name: string, n: number): BenchmarkAdapter {
|
|
11
11
|
const tasks: BenchTask[] = Array.from({ length: n }, (_, i) => ({
|
|
@@ -47,6 +47,7 @@ const shot: BenchShot = async ({ adapter, task, cell }) => {
|
|
|
47
47
|
}
|
|
48
48
|
|
|
49
49
|
async function main(): Promise<void> {
|
|
50
|
+
await managedExecutionProof()
|
|
50
51
|
// Matrix: 2 benchmarks × 3 cells × 4 tasks = 24 shots.
|
|
51
52
|
const report = await runBenchmarks({
|
|
52
53
|
benchmarks: ['alpha', 'beta'],
|
|
@@ -77,6 +78,22 @@ async function main(): Promise<void> {
|
|
|
77
78
|
assert.equal(judgeFailure.perTask[0]?.ok, false)
|
|
78
79
|
assert.equal(judgeFailure.perTask[0]?.artifact, 'PATCH', 'a judge outage retains the completed agent artifact')
|
|
79
80
|
assert.equal(judgeFailure.perTask[0]?.usage?.costUsd, 0.04, 'a judge outage retains already incurred usage')
|
|
81
|
+
assert.equal(judgeFailure.perTask[0]?.measurement, 'unavailable')
|
|
82
|
+
assert.deepEqual(judgeFailure.perTask[0]?.execution, { phase: 'started', terminalOutcome: 'succeeded' })
|
|
83
|
+
|
|
84
|
+
const failedExecution = await runBenchmarks({
|
|
85
|
+
benchmarks: ['alpha'], cells: [{ label: 'paid-failure', model: 'm' }],
|
|
86
|
+
routerBaseUrl: 'x', routerKey: 'x', n: 1, resolveAdapter: resolveStub,
|
|
87
|
+
runShot: async () => ({
|
|
88
|
+
artifact: '', ok: false, artifactAvailable: true,
|
|
89
|
+
execution: { phase: 'started', terminalOutcome: 'failed' },
|
|
90
|
+
usage: { input: 23, output: 7, costUsd: 0.04 },
|
|
91
|
+
}),
|
|
92
|
+
})
|
|
93
|
+
assert.equal(failedExecution.perTask[0]?.measurement, 'available')
|
|
94
|
+
assert.equal(failedExecution.perTask[0]?.resolved, false)
|
|
95
|
+
assert.equal(failedExecution.rows[0]?.errored, 0, 'a measured paid failure stays in the comparison')
|
|
96
|
+
assert.equal(failedExecution.perTask[0]?.usage?.input, 23)
|
|
80
97
|
|
|
81
98
|
const controller = new AbortController()
|
|
82
99
|
let started = 0
|
|
@@ -93,6 +110,7 @@ async function main(): Promise<void> {
|
|
|
93
110
|
})
|
|
94
111
|
assert.equal(started, 1, 'cancellation prevents every queued model call')
|
|
95
112
|
assert.equal(cancelled.perTask.length, 4, 'cancelled work remains visible')
|
|
113
|
+
assert.equal(cancelled.perTask.filter((row) => row.execution?.phase === 'not-started').length, 3)
|
|
96
114
|
assert.equal(cancelled.perTask[0]?.usage?.costUsd, 0.01)
|
|
97
115
|
|
|
98
116
|
const row = (b: string, c: string) => report.rows.find((r) => r.benchmark === b && r.cell === c)!
|
|
@@ -159,6 +177,20 @@ async function main(): Promise<void> {
|
|
|
159
177
|
input: 11, output: 3, costUsd: 0.02, tokensKnown: false, usdKnown: false,
|
|
160
178
|
}, 'an unreported retry preserves the measured floor without claiming complete accounting')
|
|
161
179
|
|
|
180
|
+
const measuredFailedRetry = await runBenchmarks({
|
|
181
|
+
benchmarks: ['alpha'], cells: [{ label: 'failed-retries', model: 'm' }],
|
|
182
|
+
routerBaseUrl: 'x', routerKey: 'x', resolveAdapter: resolveStub, n: 1, loopAttempts: 2,
|
|
183
|
+
runShot: async ({ attempt }) => ({
|
|
184
|
+
artifact: attempt === 1 ? '' : 'WRONG', ok: false, artifactAvailable: attempt === 2,
|
|
185
|
+
execution: { phase: 'started', terminalOutcome: 'failed' },
|
|
186
|
+
usage: { input: 11, output: 3, costUsd: 0.02 },
|
|
187
|
+
}),
|
|
188
|
+
})
|
|
189
|
+
assert.equal(measuredFailedRetry.perTask[0]?.artifact, 'WRONG')
|
|
190
|
+
assert.equal(measuredFailedRetry.perTask[0]?.measurement, 'available')
|
|
191
|
+
assert.equal(measuredFailedRetry.rows[0]?.errored, 0, 'a measured failed retry stays in the comparison')
|
|
192
|
+
assert.deepEqual(measuredFailedRetry.perTask[0]?.usage, { input: 22, output: 6, costUsd: 0.04 })
|
|
193
|
+
|
|
162
194
|
let validAttempts = 0
|
|
163
195
|
const validRetry = await runBenchmarks({
|
|
164
196
|
benchmarks: ['alpha'], cells: [{ label: 'retrying', model: 'm' }],
|
|
@@ -212,6 +244,10 @@ async function main(): Promise<void> {
|
|
|
212
244
|
const order: string[] = []
|
|
213
245
|
let createdOptions: unknown
|
|
214
246
|
let controlCredential: string | undefined
|
|
247
|
+
let completed = true
|
|
248
|
+
let text = 'fallback text'
|
|
249
|
+
let streamThrows = false
|
|
250
|
+
let setupFails = false
|
|
215
251
|
const fakeClient = {
|
|
216
252
|
async create(options: unknown) {
|
|
217
253
|
createdOptions = options
|
|
@@ -219,13 +255,14 @@ async function main(): Promise<void> {
|
|
|
219
255
|
id: 'box-default-shot',
|
|
220
256
|
async exec(command: string, options?: { sessionId?: string }) {
|
|
221
257
|
order.push(`exec:${command}:streams=${order.filter((x) => x.startsWith('stream:')).length}:session=${options?.sessionId ? 'yes' : 'no'}`)
|
|
222
|
-
return { exitCode: 0, stdout: command === 'extract-patch' ? 'PATCH' : '', stderr: '' }
|
|
258
|
+
return { exitCode: setupFails && command === 'setup-repo' ? 1 : 0, stdout: command === 'extract-patch' ? 'PATCH' : '', stderr: '' }
|
|
223
259
|
},
|
|
224
260
|
async *streamPrompt(_prompt: string, options?: { sessionId?: string }) {
|
|
225
261
|
order.push(`stream:session=${options?.sessionId ? 'yes' : 'no'}`)
|
|
226
262
|
yield { type: 'llm_call', data: { tokensIn: 23, tokensOut: 7, costUsd: 0.04 } }
|
|
227
|
-
|
|
228
|
-
yield { type: '
|
|
263
|
+
if (streamThrows) throw new Error('stream disconnected')
|
|
264
|
+
yield { type: 'result', data: { finalText: text, success: completed, status: completed ? 'success' : 'failed' } }
|
|
265
|
+
yield { type: 'done', data: { outcome: { type: completed ? 'completed' : 'failed' } } }
|
|
229
266
|
},
|
|
230
267
|
async delete() {
|
|
231
268
|
order.push('delete')
|
|
@@ -260,6 +297,8 @@ async function main(): Promise<void> {
|
|
|
260
297
|
assert.equal(boxy.rows[0]!.resolveRate, 1, 'boxExtract artifact is judged instead of fallback text')
|
|
261
298
|
assert.equal(boxy.perTask[0]?.artifact, 'PATCH')
|
|
262
299
|
assert.deepEqual(boxy.perTask[0]?.usage, { input: 23, output: 7, costUsd: 0.04 })
|
|
300
|
+
assert.deepEqual(boxy.perTask[0]?.execution, { phase: 'started', terminalOutcome: 'succeeded' })
|
|
301
|
+
assert.equal(boxy.perTask[0]?.measurement, 'available')
|
|
263
302
|
assert.equal(controlCredential, 'sandbox-control-token', 'inference grant never authorizes sandbox control')
|
|
264
303
|
assert.equal((createdOptions as { backend: { model: { apiKey: string } } }).backend.model.apiKey, 'model-grant-token')
|
|
265
304
|
assert.deepEqual(
|
|
@@ -267,6 +306,49 @@ async function main(): Promise<void> {
|
|
|
267
306
|
['exec:setup-repo:streams=0:session=yes', 'stream:session=yes', 'exec:extract-patch:streams=1:session=yes'],
|
|
268
307
|
'setup runs before the prompt stream, extract runs after the prompt stream, both in the same session',
|
|
269
308
|
)
|
|
309
|
+
for (const failure of ['stream', 'parser'] as const) {
|
|
310
|
+
streamThrows = failure === 'stream'
|
|
311
|
+
const failed = await runBenchmarks({
|
|
312
|
+
benchmarks: ['boxy'], cells: [{ label: failure, model: 'm', backend: 'sandbox' }],
|
|
313
|
+
routerBaseUrl: 'x', routerKey: 'x',
|
|
314
|
+
resolveAdapter: () => ({ ...boxAdapter, output: { parse: () => { throw new Error('parser failed') } } }),
|
|
315
|
+
resolveClient: () => fakeClient as never,
|
|
316
|
+
})
|
|
317
|
+
assert.equal(failed.perTask[0]?.measurement, 'unavailable')
|
|
318
|
+
assert.equal(failed.perTask[0]?.execution?.phase, 'started')
|
|
319
|
+
assert.deepEqual(failed.perTask[0]?.usage, {
|
|
320
|
+
input: 23, output: 7, costUsd: 0.04, tokensKnown: false, usdKnown: false,
|
|
321
|
+
}, `${failure} failure retains paid receipts without claiming complete accounting`)
|
|
322
|
+
assert.equal(failed.perTask[0]?.events?.length, failure === 'stream' ? 1 : 3, 'observed events are retained once')
|
|
323
|
+
assert.match(failed.perTask[0]?.detail ?? '', failure === 'stream' ? /stream disconnected/ : /parser failed/)
|
|
324
|
+
}
|
|
325
|
+
streamThrows = false
|
|
326
|
+
setupFails = true
|
|
327
|
+
const setupFailure = await runBenchmarks({
|
|
328
|
+
benchmarks: ['boxy'], cells: [{ label: 'setup-failure', model: 'm', backend: 'sandbox' }],
|
|
329
|
+
routerBaseUrl: 'x', routerKey: 'x', resolveAdapter: () => boxAdapter,
|
|
330
|
+
resolveClient: () => fakeClient as never,
|
|
331
|
+
})
|
|
332
|
+
assert.deepEqual(setupFailure.perTask[0]?.execution, { phase: 'unknown', terminalOutcome: 'unknown' })
|
|
333
|
+
assert.equal(setupFailure.perTask[0]?.measurement, 'unavailable')
|
|
334
|
+
assert.equal(setupFailure.perTask[0]?.usage?.tokensKnown, false, 'no observed receipt does not prove zero usage')
|
|
335
|
+
assert.equal(setupFailure.perTask[0]?.usage?.usdKnown, false)
|
|
336
|
+
assert.equal(setupFailure.perTask[0]?.events?.length, 0)
|
|
337
|
+
setupFails = false
|
|
338
|
+
for (const success of [false, true]) {
|
|
339
|
+
completed = success
|
|
340
|
+
text = ''
|
|
341
|
+
const empty = await runBenchmarks({
|
|
342
|
+
benchmarks: ['alpha'], cells: [{ label: 'empty', model: 'm', backend: 'sandbox' }],
|
|
343
|
+
routerBaseUrl: 'x', routerKey: 'x', n: 1, resolveAdapter: resolveStub,
|
|
344
|
+
resolveClient: () => fakeClient as never,
|
|
345
|
+
})
|
|
346
|
+
assert.equal(empty.perTask[0]?.ok, false)
|
|
347
|
+
assert.equal(empty.perTask[0]?.measurement, 'available', 'captured empty output is a measurable failure')
|
|
348
|
+
assert.equal(empty.perTask[0]?.execution?.terminalOutcome, success ? 'succeeded' : 'failed')
|
|
349
|
+
assert.equal(empty.rows[0]?.errored, 0)
|
|
350
|
+
assert.equal(empty.perTask[0]?.usage?.input, 23)
|
|
351
|
+
}
|
|
270
352
|
}
|
|
271
353
|
|
|
272
354
|
// An unavailable benchmark (preflight throws) is skipped, not fatal; the sweep still runs the rest.
|
|
@@ -310,4 +392,180 @@ async function main(): Promise<void> {
|
|
|
310
392
|
console.log('run-benchmarks.test: OK (24-shot matrix, subset, reps, unavailable-skip, judge self-check, guards)')
|
|
311
393
|
}
|
|
312
394
|
|
|
395
|
+
async function managedExecutionProof(): Promise<void> {
|
|
396
|
+
function fixture(options: { failure?: 'stream' | 'abort'; signal?: AbortSignal; onSecond?: () => void } = {}) {
|
|
397
|
+
const operations: string[] = []
|
|
398
|
+
const requests: Array<{ prompt: string; sessionId?: string }> = []
|
|
399
|
+
let creates = 0
|
|
400
|
+
let grades = 0
|
|
401
|
+
const client = {
|
|
402
|
+
async create() {
|
|
403
|
+
creates += 1
|
|
404
|
+
let patch = 'WRONG'
|
|
405
|
+
let turns = 0
|
|
406
|
+
return {
|
|
407
|
+
id: `managed-box-${creates}`,
|
|
408
|
+
async exec(command: string, opts?: { sessionId?: string }) {
|
|
409
|
+
operations.push(command)
|
|
410
|
+
assert.ok(opts?.sessionId, 'working checks and extraction address the worker session')
|
|
411
|
+
return { exitCode: 0, stdout: command === 'extract' ? patch : 'working check: repair the missing branch', stderr: '' }
|
|
412
|
+
},
|
|
413
|
+
async *streamPrompt(prompt: string, opts?: { sessionId?: string; signal?: AbortSignal }) {
|
|
414
|
+
turns += 1
|
|
415
|
+
requests.push({ prompt, sessionId: opts?.sessionId })
|
|
416
|
+
operations.push(`prompt:${turns}`)
|
|
417
|
+
// A repeated receipt id across prompts must not collapse paid calls.
|
|
418
|
+
yield { type: 'llm_call', data: { id: 'same-receipt-id', tokensIn: 11, tokensOut: 3, costUsd: 0.02 } }
|
|
419
|
+
if (turns === 2) {
|
|
420
|
+
options.onSecond?.()
|
|
421
|
+
if (options.failure === 'stream') throw new Error('second prompt disconnected')
|
|
422
|
+
if (options.failure === 'abort') {
|
|
423
|
+
assert.equal(opts?.signal?.aborted, true)
|
|
424
|
+
throw Object.assign(new Error('cancelled second prompt'), { name: 'AbortError' })
|
|
425
|
+
}
|
|
426
|
+
}
|
|
427
|
+
if (prompt.includes('repair the missing branch')) patch = 'PATCH'
|
|
428
|
+
yield { type: 'result', data: { finalText: patch, success: true, status: 'success' } }
|
|
429
|
+
yield { type: 'done', data: { outcome: { type: 'completed' } } }
|
|
430
|
+
},
|
|
431
|
+
async delete() { operations.push('delete') },
|
|
432
|
+
}
|
|
433
|
+
},
|
|
434
|
+
async criuStatus() { return { available: false } },
|
|
435
|
+
}
|
|
436
|
+
const adapter: BenchmarkAdapter = {
|
|
437
|
+
name: 'managed',
|
|
438
|
+
preflight: async () => {},
|
|
439
|
+
loadTasks: async () => [{ id: 'task', prompt: 'fix the branch', metadata: { gold: 'PRIVATE FINAL ORACLE' } }],
|
|
440
|
+
boxSetup: () => ({ command: 'setup' }),
|
|
441
|
+
boxExtract: () => ({ command: 'extract' }),
|
|
442
|
+
goldArtifact: async () => 'PATCH',
|
|
443
|
+
judge: async (_task, artifact) => {
|
|
444
|
+
grades += 1
|
|
445
|
+
operations.push('judge')
|
|
446
|
+
return { resolved: artifact === 'PATCH', score: artifact === 'PATCH' ? 1 : 0 }
|
|
447
|
+
},
|
|
448
|
+
}
|
|
449
|
+
return {
|
|
450
|
+
operations, requests, get creates() { return creates }, get grades() { return grades },
|
|
451
|
+
run: (execute?: BenchExecution, loopAttempts = 1) => runBenchmarks({
|
|
452
|
+
benchmarks: ['managed'], cells: [{ label: 'worker', model: 'model', backend: 'sandbox' }],
|
|
453
|
+
routerBaseUrl: 'unused', routerKey: 'unused', verifyJudge: false, loopAttempts,
|
|
454
|
+
...(options.signal ? { signal: options.signal } : {}),
|
|
455
|
+
...(execute ? { execute } : {}),
|
|
456
|
+
resolveAdapter: () => adapter, resolveClient: () => client as never,
|
|
457
|
+
}),
|
|
458
|
+
}
|
|
459
|
+
}
|
|
460
|
+
const correct: BenchExecution = async (context) => {
|
|
461
|
+
assert.deepEqual(Object.keys(context).sort(), ['attempt', 'benchmark', 'profile', 'prompt', 'run', 'signal', 'taskId'])
|
|
462
|
+
assert.equal('close' in context.run, false)
|
|
463
|
+
assert.equal(context.profile.model?.default, 'model')
|
|
464
|
+
const first = await context.run.start(context.prompt)
|
|
465
|
+
assert.equal(first.out, 'WRONG')
|
|
466
|
+
const feedback = await context.run.box.exec('working-check', { sessionId: context.run.sessionId })
|
|
467
|
+
await context.run.resume(feedback.stdout)
|
|
468
|
+
}
|
|
469
|
+
const enabled = fixture()
|
|
470
|
+
const report = await enabled.run(correct)
|
|
471
|
+
assert.equal(report.perTask[0]?.resolved, true)
|
|
472
|
+
assert.deepEqual(enabled.operations, ['setup', 'prompt:1', 'working-check', 'prompt:2', 'extract', 'delete', 'judge'])
|
|
473
|
+
assert.equal(enabled.creates, 1)
|
|
474
|
+
assert.equal(enabled.grades, 1)
|
|
475
|
+
assert.equal(enabled.requests[0]?.sessionId, enabled.requests[1]?.sessionId)
|
|
476
|
+
assert.equal(enabled.requests[1]?.prompt, 'working check: repair the missing branch')
|
|
477
|
+
assert.deepEqual(report.perTask[0]?.usage, { input: 22, output: 6, costUsd: 0.04 })
|
|
478
|
+
assert.deepEqual(report.perTask[0]?.prompts?.map((p) => [p.attempt, p.index, p.method, p.prompt]), [
|
|
479
|
+
[1, 0, 'start', 'fix the branch'], [1, 1, 'resume', 'working check: repair the missing branch'],
|
|
480
|
+
])
|
|
481
|
+
assert.equal(report.perTask[0]?.events?.length, 6)
|
|
482
|
+
const disabled = fixture()
|
|
483
|
+
const withheld = await disabled.run(async ({ run, prompt }) => {
|
|
484
|
+
await run.start(prompt)
|
|
485
|
+
await run.resume('try again without a correction')
|
|
486
|
+
})
|
|
487
|
+
assert.equal(withheld.perTask[0]?.resolved, false, 'withholding the correction prevents the scripted repair')
|
|
488
|
+
assert.deepEqual(withheld.perTask[0]?.usage, report.perTask[0]?.usage, 'control spends the same scripted resources')
|
|
489
|
+
const defaultRun = await fixture().run()
|
|
490
|
+
assert.equal(defaultRun.perTask[0]?.prompts?.length, 1)
|
|
491
|
+
assert.deepEqual(defaultRun.perTask[0]?.usage, { input: 11, output: 3, costUsd: 0.02 })
|
|
492
|
+
|
|
493
|
+
for (const failure of ['stream', 'abort'] as const) {
|
|
494
|
+
const controller = new AbortController()
|
|
495
|
+
const broken = fixture({ failure, signal: controller.signal, onSecond: () => { if (failure === 'abort') controller.abort() } })
|
|
496
|
+
const failed = await broken.run(correct)
|
|
497
|
+
const row = failed.perTask[0]!
|
|
498
|
+
assert.equal(row.ok, false)
|
|
499
|
+
assert.equal(row.measurement, 'unavailable')
|
|
500
|
+
assert.equal(row.prompts?.length, 2)
|
|
501
|
+
assert.equal(row.events?.length, 4)
|
|
502
|
+
assert.deepEqual(row.usage, { input: 22, output: 6, costUsd: 0.04, tokensKnown: false, usdKnown: false })
|
|
503
|
+
assert.equal(row.prompts?.[0]?.usage.tokensKnown, undefined)
|
|
504
|
+
assert.equal(row.prompts?.[1]?.usage.tokensKnown, false)
|
|
505
|
+
assert.equal(broken.operations.filter((op) => op === 'delete').length, 1)
|
|
506
|
+
assert.equal(broken.operations.includes('extract'), false)
|
|
507
|
+
}
|
|
508
|
+
const policyFailure = fixture()
|
|
509
|
+
const policyFailed = await policyFailure.run(async ({ run, prompt }) => {
|
|
510
|
+
await run.start(prompt)
|
|
511
|
+
throw new Error('policy failed after paid work')
|
|
512
|
+
})
|
|
513
|
+
assert.equal(policyFailed.perTask[0]?.ok, false)
|
|
514
|
+
assert.deepEqual(policyFailed.perTask[0]?.usage, { input: 11, output: 3, costUsd: 0.02 })
|
|
515
|
+
assert.match(policyFailed.perTask[0]?.detail ?? '', /policy failed/)
|
|
516
|
+
assert.equal(policyFailure.operations.filter((op) => op === 'delete').length, 1)
|
|
517
|
+
|
|
518
|
+
const policyAbortController = new AbortController()
|
|
519
|
+
const policyAbort = fixture({ signal: policyAbortController.signal })
|
|
520
|
+
const abortedPolicy = await policyAbort.run(async ({ run, prompt }) => {
|
|
521
|
+
await run.start(prompt)
|
|
522
|
+
policyAbortController.abort()
|
|
523
|
+
await new Promise<void>(() => {})
|
|
524
|
+
})
|
|
525
|
+
assert.equal(abortedPolicy.perTask[0]?.ok, false, 'cancellation stops waiting for policy work')
|
|
526
|
+
assert.equal(abortedPolicy.perTask[0]?.prompts?.length, 1)
|
|
527
|
+
assert.deepEqual(abortedPolicy.perTask[0]?.usage, { input: 11, output: 3, costUsd: 0.02 })
|
|
528
|
+
assert.equal(policyAbort.operations.filter((op) => op === 'delete').length, 1)
|
|
529
|
+
|
|
530
|
+
const looped = fixture()
|
|
531
|
+
const retried = await looped.run(async ({ run, prompt, attempt }) => {
|
|
532
|
+
await run.start(prompt)
|
|
533
|
+
await run.resume(attempt === 2 ? 'repair the missing branch' : 'check again')
|
|
534
|
+
}, 2)
|
|
535
|
+
assert.equal(retried.perTask[0]?.resolved, true)
|
|
536
|
+
assert.equal(looped.creates, 2)
|
|
537
|
+
assert.deepEqual(retried.perTask[0]?.prompts?.map((p) => [p.attempt, p.index]), [[1, 0], [1, 1], [2, 0], [2, 1]])
|
|
538
|
+
assert.deepEqual(retried.perTask[0]?.usage, { input: 44, output: 12, costUsd: 0.08 })
|
|
539
|
+
assert.equal(looped.operations.filter((op) => op === 'setup').length, 2)
|
|
540
|
+
assert.equal(looped.operations.filter((op) => op === 'extract').length, 2)
|
|
541
|
+
assert.equal(looped.operations.filter((op) => op === 'delete').length, 2)
|
|
542
|
+
|
|
543
|
+
const unawaited = fixture()
|
|
544
|
+
const awaitedByOwner = await unawaited.run(async ({ run, prompt }) => { void run.start(prompt) })
|
|
545
|
+
assert.equal(awaitedByOwner.perTask[0]?.prompts?.length, 1)
|
|
546
|
+
assert.deepEqual(unawaited.operations, ['setup', 'prompt:1', 'extract', 'delete', 'judge'])
|
|
547
|
+
const overlapping = fixture()
|
|
548
|
+
const overlap = await overlapping.run(async ({ run, prompt }) => {
|
|
549
|
+
const first = run.start(prompt)
|
|
550
|
+
assert.throws(() => run.resume('overlap'), /sequential/)
|
|
551
|
+
await first
|
|
552
|
+
})
|
|
553
|
+
assert.equal(overlap.perTask[0]?.ok, false)
|
|
554
|
+
assert.equal(overlapping.operations.includes('extract'), false)
|
|
555
|
+
let retained: BenchExecutionContext['run'] | undefined
|
|
556
|
+
await fixture().run(async ({ run, prompt }) => { retained = run; await run.start(prompt) })
|
|
557
|
+
assert.throws(() => retained!.resume('too late'), /settled/)
|
|
558
|
+
const skipped = await fixture().run(async () => {})
|
|
559
|
+
assert.equal(skipped.perTask[0]?.ok, false)
|
|
560
|
+
assert.equal(skipped.perTask[0]?.prompts?.length, 0)
|
|
561
|
+
const passthrough: BenchExecution = async () => {}
|
|
562
|
+
let received: BenchExecution | undefined
|
|
563
|
+
await runBenchmarks({
|
|
564
|
+
benchmarks: ['alpha'], cells: [{ label: 'custom', model: 'm' }], n: 1,
|
|
565
|
+
routerBaseUrl: 'unused', routerKey: 'unused', resolveAdapter: resolveStub,
|
|
566
|
+
execute: passthrough, runShot: async ({ execute }) => { received = execute; return { artifact: '', ok: false } },
|
|
567
|
+
})
|
|
568
|
+
assert.equal(received, passthrough, 'custom shots choose how to consume the callback')
|
|
569
|
+
}
|
|
570
|
+
|
|
313
571
|
void main()
|