@tangle-network/agent-bench 0.9.4 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/HARNESS.md +10 -10
- package/dist/index.d.ts +12 -3
- package/dist/index.js +55 -9
- package/dist/index.js.map +1 -1
- package/package.json +5 -5
- package/scripts/run-package-tests.mjs +12 -2
- package/scripts/run-package-tests.test.mjs +26 -1
- package/src/run-benchmarks.test.mts +84 -3
- package/src/run-benchmarks.ts +52 -9
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-bench",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.10.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Benchmark adapters and execution for agent-runtime across coding, tool-use, RAG, memory, browser, and terminal tasks.",
|
|
6
6
|
"repository": {
|
|
@@ -25,11 +25,11 @@
|
|
|
25
25
|
}
|
|
26
26
|
},
|
|
27
27
|
"dependencies": {
|
|
28
|
-
"@tangle-network/agent-eval": ">=0.
|
|
29
|
-
"@tangle-network/agent-interface": "^2.
|
|
30
|
-
"@tangle-network/agent-knowledge": "^15.0.
|
|
28
|
+
"@tangle-network/agent-eval": ">=0.178.0 <0.179.0",
|
|
29
|
+
"@tangle-network/agent-interface": "^2.6.0",
|
|
30
|
+
"@tangle-network/agent-knowledge": "^15.0.1",
|
|
31
31
|
"@tangle-network/sandbox": ">=0.36.4 <0.38.0",
|
|
32
|
-
"@tangle-network/agent-runtime": "^0.
|
|
32
|
+
"@tangle-network/agent-runtime": "^0.203.0"
|
|
33
33
|
},
|
|
34
34
|
"devDependencies": {
|
|
35
35
|
"@arethetypeswrong/cli": "0.18.5",
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { execFile } from 'node:child_process'
|
|
2
|
-
import { access, readdir, readFile } from 'node:fs/promises'
|
|
2
|
+
import { access, readdir, readFile, realpath } from 'node:fs/promises'
|
|
3
|
+
import { tmpdir } from 'node:os'
|
|
3
4
|
import path from 'node:path'
|
|
4
5
|
import { fileURLToPath } from 'node:url'
|
|
5
6
|
import { promisify } from 'node:util'
|
|
@@ -64,6 +65,15 @@ export async function run(command, args, env = process.env) {
|
|
|
64
65
|
}
|
|
65
66
|
}
|
|
66
67
|
|
|
68
|
+
export async function runPythonTests(python, env = process.env) {
|
|
69
|
+
// Fixture roots must be physical paths; the production boundary rejects symlinked ancestors.
|
|
70
|
+
const physicalTemp = await realpath(env.TMPDIR || tmpdir())
|
|
71
|
+
await run(python, ['-m', 'unittest', 'discover', '-s', 'pier_agents', '-p', '*_test.py'], {
|
|
72
|
+
...env,
|
|
73
|
+
TMPDIR: physicalTemp,
|
|
74
|
+
})
|
|
75
|
+
}
|
|
76
|
+
|
|
67
77
|
async function main() {
|
|
68
78
|
const python = path.join(benchDir, '.venv', 'bin', 'python')
|
|
69
79
|
try {
|
|
@@ -101,7 +111,7 @@ async function main() {
|
|
|
101
111
|
await run('npx', ['vitest', 'run', ...vitestTests.map((file) => path.relative(benchDir, file))])
|
|
102
112
|
}
|
|
103
113
|
|
|
104
|
-
await
|
|
114
|
+
await runPythonTests(python)
|
|
105
115
|
|
|
106
116
|
console.log(
|
|
107
117
|
`package tests passed: ${tests.length}/${tests.length} TypeScript files (${nodeTests.length} node:test + ${vitestTests.length} vitest) + Pier bridge`,
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
import assert from 'node:assert/strict'
|
|
2
2
|
import { test } from 'node:test'
|
|
3
|
-
import {
|
|
3
|
+
import { mkdtemp, mkdir, readFile, realpath, rm, symlink, writeFile } from 'node:fs/promises'
|
|
4
|
+
import { tmpdir } from 'node:os'
|
|
5
|
+
import path from 'node:path'
|
|
6
|
+
import { packageNodeTestArgs, resolvePackageTestTimeoutMs, run, runPythonTests } from './run-package-tests.mjs'
|
|
4
7
|
|
|
5
8
|
test('package test concurrency reaches Node without changing selected files', () => {
|
|
6
9
|
const files = ['src/first.test.mts', 'src/second.test.ts']
|
|
@@ -45,3 +48,25 @@ test('package test timeout reaches the child process', async () => {
|
|
|
45
48
|
)
|
|
46
49
|
assert.ok(Date.now() - startedAt < 2_000)
|
|
47
50
|
})
|
|
51
|
+
|
|
52
|
+
test('Python fixtures use physical temporary paths without relaxing symlink guards', async () => {
|
|
53
|
+
const root = await mkdtemp(path.join(await realpath(tmpdir()), 'bench-python-temp-'))
|
|
54
|
+
try {
|
|
55
|
+
const target = path.join(root, 'physical')
|
|
56
|
+
const linked = path.join(root, 'linked')
|
|
57
|
+
const recorded = path.join(root, 'recorded.json')
|
|
58
|
+
const launcher = path.join(root, 'python-fixture')
|
|
59
|
+
await mkdir(target)
|
|
60
|
+
await symlink(target, linked, 'dir')
|
|
61
|
+
await writeFile(launcher, `#!${process.execPath}
|
|
62
|
+
const fs = require('node:fs'); fs.writeFileSync(process.env.RECORDED, JSON.stringify({ temporary: process.env.TMPDIR, args: process.argv.slice(2) }))
|
|
63
|
+
`, { mode: 0o755 })
|
|
64
|
+
await runPythonTests(launcher, { ...process.env, TMPDIR: linked, RECORDED: recorded })
|
|
65
|
+
assert.deepEqual(JSON.parse(await readFile(recorded, 'utf8')), {
|
|
66
|
+
temporary: target,
|
|
67
|
+
args: ['-m', 'unittest', 'discover', '-s', 'pier_agents', '-p', '*_test.py'],
|
|
68
|
+
})
|
|
69
|
+
} finally {
|
|
70
|
+
await rm(root, { recursive: true, force: true })
|
|
71
|
+
}
|
|
72
|
+
})
|
|
@@ -77,6 +77,22 @@ async function main(): Promise<void> {
|
|
|
77
77
|
assert.equal(judgeFailure.perTask[0]?.ok, false)
|
|
78
78
|
assert.equal(judgeFailure.perTask[0]?.artifact, 'PATCH', 'a judge outage retains the completed agent artifact')
|
|
79
79
|
assert.equal(judgeFailure.perTask[0]?.usage?.costUsd, 0.04, 'a judge outage retains already incurred usage')
|
|
80
|
+
assert.equal(judgeFailure.perTask[0]?.measurement, 'unavailable')
|
|
81
|
+
assert.deepEqual(judgeFailure.perTask[0]?.execution, { phase: 'started', terminalOutcome: 'succeeded' })
|
|
82
|
+
|
|
83
|
+
const failedExecution = await runBenchmarks({
|
|
84
|
+
benchmarks: ['alpha'], cells: [{ label: 'paid-failure', model: 'm' }],
|
|
85
|
+
routerBaseUrl: 'x', routerKey: 'x', n: 1, resolveAdapter: resolveStub,
|
|
86
|
+
runShot: async () => ({
|
|
87
|
+
artifact: '', ok: false, artifactAvailable: true,
|
|
88
|
+
execution: { phase: 'started', terminalOutcome: 'failed' },
|
|
89
|
+
usage: { input: 23, output: 7, costUsd: 0.04 },
|
|
90
|
+
}),
|
|
91
|
+
})
|
|
92
|
+
assert.equal(failedExecution.perTask[0]?.measurement, 'available')
|
|
93
|
+
assert.equal(failedExecution.perTask[0]?.resolved, false)
|
|
94
|
+
assert.equal(failedExecution.rows[0]?.errored, 0, 'a measured paid failure stays in the comparison')
|
|
95
|
+
assert.equal(failedExecution.perTask[0]?.usage?.input, 23)
|
|
80
96
|
|
|
81
97
|
const controller = new AbortController()
|
|
82
98
|
let started = 0
|
|
@@ -93,6 +109,7 @@ async function main(): Promise<void> {
|
|
|
93
109
|
})
|
|
94
110
|
assert.equal(started, 1, 'cancellation prevents every queued model call')
|
|
95
111
|
assert.equal(cancelled.perTask.length, 4, 'cancelled work remains visible')
|
|
112
|
+
assert.equal(cancelled.perTask.filter((row) => row.execution?.phase === 'not-started').length, 3)
|
|
96
113
|
assert.equal(cancelled.perTask[0]?.usage?.costUsd, 0.01)
|
|
97
114
|
|
|
98
115
|
const row = (b: string, c: string) => report.rows.find((r) => r.benchmark === b && r.cell === c)!
|
|
@@ -159,6 +176,20 @@ async function main(): Promise<void> {
|
|
|
159
176
|
input: 11, output: 3, costUsd: 0.02, tokensKnown: false, usdKnown: false,
|
|
160
177
|
}, 'an unreported retry preserves the measured floor without claiming complete accounting')
|
|
161
178
|
|
|
179
|
+
const measuredFailedRetry = await runBenchmarks({
|
|
180
|
+
benchmarks: ['alpha'], cells: [{ label: 'failed-retries', model: 'm' }],
|
|
181
|
+
routerBaseUrl: 'x', routerKey: 'x', resolveAdapter: resolveStub, n: 1, loopAttempts: 2,
|
|
182
|
+
runShot: async ({ attempt }) => ({
|
|
183
|
+
artifact: attempt === 1 ? '' : 'WRONG', ok: false, artifactAvailable: attempt === 2,
|
|
184
|
+
execution: { phase: 'started', terminalOutcome: 'failed' },
|
|
185
|
+
usage: { input: 11, output: 3, costUsd: 0.02 },
|
|
186
|
+
}),
|
|
187
|
+
})
|
|
188
|
+
assert.equal(measuredFailedRetry.perTask[0]?.artifact, 'WRONG')
|
|
189
|
+
assert.equal(measuredFailedRetry.perTask[0]?.measurement, 'available')
|
|
190
|
+
assert.equal(measuredFailedRetry.rows[0]?.errored, 0, 'a measured failed retry stays in the comparison')
|
|
191
|
+
assert.deepEqual(measuredFailedRetry.perTask[0]?.usage, { input: 22, output: 6, costUsd: 0.04 })
|
|
192
|
+
|
|
162
193
|
let validAttempts = 0
|
|
163
194
|
const validRetry = await runBenchmarks({
|
|
164
195
|
benchmarks: ['alpha'], cells: [{ label: 'retrying', model: 'm' }],
|
|
@@ -212,6 +243,10 @@ async function main(): Promise<void> {
|
|
|
212
243
|
const order: string[] = []
|
|
213
244
|
let createdOptions: unknown
|
|
214
245
|
let controlCredential: string | undefined
|
|
246
|
+
let completed = true
|
|
247
|
+
let text = 'fallback text'
|
|
248
|
+
let streamThrows = false
|
|
249
|
+
let setupFails = false
|
|
215
250
|
const fakeClient = {
|
|
216
251
|
async create(options: unknown) {
|
|
217
252
|
createdOptions = options
|
|
@@ -219,13 +254,14 @@ async function main(): Promise<void> {
|
|
|
219
254
|
id: 'box-default-shot',
|
|
220
255
|
async exec(command: string, options?: { sessionId?: string }) {
|
|
221
256
|
order.push(`exec:${command}:streams=${order.filter((x) => x.startsWith('stream:')).length}:session=${options?.sessionId ? 'yes' : 'no'}`)
|
|
222
|
-
return { exitCode: 0, stdout: command === 'extract-patch' ? 'PATCH' : '', stderr: '' }
|
|
257
|
+
return { exitCode: setupFails && command === 'setup-repo' ? 1 : 0, stdout: command === 'extract-patch' ? 'PATCH' : '', stderr: '' }
|
|
223
258
|
},
|
|
224
259
|
async *streamPrompt(_prompt: string, options?: { sessionId?: string }) {
|
|
225
260
|
order.push(`stream:session=${options?.sessionId ? 'yes' : 'no'}`)
|
|
226
261
|
yield { type: 'llm_call', data: { tokensIn: 23, tokensOut: 7, costUsd: 0.04 } }
|
|
227
|
-
|
|
228
|
-
yield { type: '
|
|
262
|
+
if (streamThrows) throw new Error('stream disconnected')
|
|
263
|
+
yield { type: 'result', data: { finalText: text, success: completed, status: completed ? 'success' : 'failed' } }
|
|
264
|
+
yield { type: 'done', data: { outcome: { type: completed ? 'completed' : 'failed' } } }
|
|
229
265
|
},
|
|
230
266
|
async delete() {
|
|
231
267
|
order.push('delete')
|
|
@@ -260,6 +296,8 @@ async function main(): Promise<void> {
|
|
|
260
296
|
assert.equal(boxy.rows[0]!.resolveRate, 1, 'boxExtract artifact is judged instead of fallback text')
|
|
261
297
|
assert.equal(boxy.perTask[0]?.artifact, 'PATCH')
|
|
262
298
|
assert.deepEqual(boxy.perTask[0]?.usage, { input: 23, output: 7, costUsd: 0.04 })
|
|
299
|
+
assert.deepEqual(boxy.perTask[0]?.execution, { phase: 'started', terminalOutcome: 'succeeded' })
|
|
300
|
+
assert.equal(boxy.perTask[0]?.measurement, 'available')
|
|
263
301
|
assert.equal(controlCredential, 'sandbox-control-token', 'inference grant never authorizes sandbox control')
|
|
264
302
|
assert.equal((createdOptions as { backend: { model: { apiKey: string } } }).backend.model.apiKey, 'model-grant-token')
|
|
265
303
|
assert.deepEqual(
|
|
@@ -267,6 +305,49 @@ async function main(): Promise<void> {
|
|
|
267
305
|
['exec:setup-repo:streams=0:session=yes', 'stream:session=yes', 'exec:extract-patch:streams=1:session=yes'],
|
|
268
306
|
'setup runs before the prompt stream, extract runs after the prompt stream, both in the same session',
|
|
269
307
|
)
|
|
308
|
+
for (const failure of ['stream', 'parser'] as const) {
|
|
309
|
+
streamThrows = failure === 'stream'
|
|
310
|
+
const failed = await runBenchmarks({
|
|
311
|
+
benchmarks: ['boxy'], cells: [{ label: failure, model: 'm', backend: 'sandbox' }],
|
|
312
|
+
routerBaseUrl: 'x', routerKey: 'x',
|
|
313
|
+
resolveAdapter: () => ({ ...boxAdapter, output: { parse: () => { throw new Error('parser failed') } } }),
|
|
314
|
+
resolveClient: () => fakeClient as never,
|
|
315
|
+
})
|
|
316
|
+
assert.equal(failed.perTask[0]?.measurement, 'unavailable')
|
|
317
|
+
assert.equal(failed.perTask[0]?.execution?.phase, 'started')
|
|
318
|
+
assert.deepEqual(failed.perTask[0]?.usage, {
|
|
319
|
+
input: 23, output: 7, costUsd: 0.04, tokensKnown: false, usdKnown: false,
|
|
320
|
+
}, `${failure} failure retains paid receipts without claiming complete accounting`)
|
|
321
|
+
assert.equal(failed.perTask[0]?.events?.length, failure === 'stream' ? 1 : 3, 'observed events are retained once')
|
|
322
|
+
assert.match(failed.perTask[0]?.detail ?? '', failure === 'stream' ? /stream disconnected/ : /parser failed/)
|
|
323
|
+
}
|
|
324
|
+
streamThrows = false
|
|
325
|
+
setupFails = true
|
|
326
|
+
const setupFailure = await runBenchmarks({
|
|
327
|
+
benchmarks: ['boxy'], cells: [{ label: 'setup-failure', model: 'm', backend: 'sandbox' }],
|
|
328
|
+
routerBaseUrl: 'x', routerKey: 'x', resolveAdapter: () => boxAdapter,
|
|
329
|
+
resolveClient: () => fakeClient as never,
|
|
330
|
+
})
|
|
331
|
+
assert.deepEqual(setupFailure.perTask[0]?.execution, { phase: 'unknown', terminalOutcome: 'unknown' })
|
|
332
|
+
assert.equal(setupFailure.perTask[0]?.measurement, 'unavailable')
|
|
333
|
+
assert.equal(setupFailure.perTask[0]?.usage?.tokensKnown, false, 'no observed receipt does not prove zero usage')
|
|
334
|
+
assert.equal(setupFailure.perTask[0]?.usage?.usdKnown, false)
|
|
335
|
+
assert.equal(setupFailure.perTask[0]?.events?.length, 0)
|
|
336
|
+
setupFails = false
|
|
337
|
+
for (const success of [false, true]) {
|
|
338
|
+
completed = success
|
|
339
|
+
text = ''
|
|
340
|
+
const empty = await runBenchmarks({
|
|
341
|
+
benchmarks: ['alpha'], cells: [{ label: 'empty', model: 'm', backend: 'sandbox' }],
|
|
342
|
+
routerBaseUrl: 'x', routerKey: 'x', n: 1, resolveAdapter: resolveStub,
|
|
343
|
+
resolveClient: () => fakeClient as never,
|
|
344
|
+
})
|
|
345
|
+
assert.equal(empty.perTask[0]?.ok, false)
|
|
346
|
+
assert.equal(empty.perTask[0]?.measurement, 'available', 'captured empty output is a measurable failure')
|
|
347
|
+
assert.equal(empty.perTask[0]?.execution?.terminalOutcome, success ? 'succeeded' : 'failed')
|
|
348
|
+
assert.equal(empty.rows[0]?.errored, 0)
|
|
349
|
+
assert.equal(empty.perTask[0]?.usage?.input, 23)
|
|
350
|
+
}
|
|
270
351
|
}
|
|
271
352
|
|
|
272
353
|
// An unavailable benchmark (preflight throws) is skipped, not fatal; the sweep still runs the rest.
|
package/src/run-benchmarks.ts
CHANGED
|
@@ -29,6 +29,7 @@
|
|
|
29
29
|
*/
|
|
30
30
|
|
|
31
31
|
import { mkdirSync, writeFileSync } from 'node:fs'
|
|
32
|
+
import type { RunTerminalOutcome } from '@tangle-network/agent-eval'
|
|
32
33
|
import type {
|
|
33
34
|
AgentProfile,
|
|
34
35
|
AgentRunSpec,
|
|
@@ -68,6 +69,13 @@ export interface BenchShotResult {
|
|
|
68
69
|
/** Provider observations, including explicit unknown counters. Omitted when the shot reports none. */
|
|
69
70
|
readonly usage?: ReturnType<typeof sumSandboxUsage>
|
|
70
71
|
readonly events?: readonly SandboxEvent[]
|
|
72
|
+
/** Observed dispatch and terminal state, independent of artifact quality. */
|
|
73
|
+
readonly execution?: {
|
|
74
|
+
readonly phase: 'not-started' | 'started' | 'unknown'
|
|
75
|
+
readonly terminalOutcome: RunTerminalOutcome
|
|
76
|
+
}
|
|
77
|
+
/** Whether the artifact was captured without a read or extraction failure. */
|
|
78
|
+
readonly artifactAvailable?: boolean
|
|
71
79
|
}
|
|
72
80
|
|
|
73
81
|
/** Runs one (adapter, task, cell) shot. Defaults to `openSandboxRun`. */
|
|
@@ -137,9 +145,11 @@ export interface BenchCellTaskResult {
|
|
|
137
145
|
readonly rep: number
|
|
138
146
|
readonly resolved: boolean
|
|
139
147
|
readonly score: number
|
|
140
|
-
/**
|
|
141
|
-
* denominator so a harness outage can't masquerade as a 0% capability result. */
|
|
148
|
+
/** Whether execution completed successfully and produced a readable, nonempty artifact. */
|
|
142
149
|
readonly ok: boolean
|
|
150
|
+
readonly execution?: BenchShotResult['execution']
|
|
151
|
+
/** Available failed attempts remain in comparisons; unavailable measurement is reported separately. */
|
|
152
|
+
readonly measurement?: 'available' | 'unavailable'
|
|
143
153
|
readonly detail?: string
|
|
144
154
|
readonly wallMs: number
|
|
145
155
|
/** Exact bytes given to the benchmark judge, retained even when judging fails. */
|
|
@@ -237,11 +247,13 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
|
|
|
237
247
|
}
|
|
238
248
|
const controller = new AbortController()
|
|
239
249
|
const timer = timeoutMs ? setTimeout(() => controller.abort(), timeoutMs) : undefined
|
|
250
|
+
const observedEvents: SandboxEvent[] = []
|
|
240
251
|
const runOptions: OpenSandboxRunOptions = {
|
|
241
252
|
agentRun,
|
|
242
253
|
signal: signal ? AbortSignal.any([controller.signal, signal]) : controller.signal,
|
|
243
254
|
runId: `bench:${adapter.name}:${task.id}:${uniq}`,
|
|
244
255
|
scenarioId: task.id,
|
|
256
|
+
onSandboxEvent: (event) => { observedEvents.push(event) },
|
|
245
257
|
}
|
|
246
258
|
const boxSetup = adapter.boxSetup
|
|
247
259
|
if (boxSetup) {
|
|
@@ -262,8 +274,15 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
|
|
|
262
274
|
let result: BenchShotResult = { artifact: '', ok: false }
|
|
263
275
|
try {
|
|
264
276
|
run = await openSandboxRun(client, runOptions, deliverable)
|
|
277
|
+
result = { ...result, execution: { phase: 'unknown', terminalOutcome: 'unknown' } }
|
|
265
278
|
const turn = await run.start(prompt ?? task.prompt)
|
|
266
|
-
result = {
|
|
279
|
+
result = {
|
|
280
|
+
artifact: '', ok: false, usage: sumSandboxUsage(turn.events), events: turn.events,
|
|
281
|
+
execution: {
|
|
282
|
+
phase: 'started',
|
|
283
|
+
terminalOutcome: turn.outcome.success ? 'succeeded' : turn.outcome.status === 'failed' ? 'failed' : 'incomplete',
|
|
284
|
+
},
|
|
285
|
+
}
|
|
267
286
|
// Event-stream deliverable (adapter.output ?? finalText) — the FALLBACK.
|
|
268
287
|
let artifact = (turn.out ?? '').trim()
|
|
269
288
|
let boxExtractError: string | undefined
|
|
@@ -336,16 +355,26 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
|
|
|
336
355
|
result = {
|
|
337
356
|
artifact,
|
|
338
357
|
ok: turn.outcome.success && artifact.length > 0 && turn.readError === undefined && boxExtractError === undefined,
|
|
358
|
+
execution: result.execution,
|
|
359
|
+
artifactAvailable: turn.readError === undefined && boxExtractError === undefined,
|
|
339
360
|
usage: result.usage,
|
|
340
361
|
events: turn.events,
|
|
341
362
|
...(detail ? { detail } : {}),
|
|
342
363
|
}
|
|
343
364
|
} catch (err) {
|
|
365
|
+
const events = err instanceof SandboxRunAbortError ? err.events : observedEvents
|
|
344
366
|
result = {
|
|
345
367
|
...result,
|
|
346
368
|
ok: false,
|
|
369
|
+
artifactAvailable: false,
|
|
370
|
+
execution: {
|
|
371
|
+
phase: events.length > 0 ? 'started' : 'unknown',
|
|
372
|
+
terminalOutcome: result.execution?.terminalOutcome ?? 'unknown',
|
|
373
|
+
},
|
|
347
374
|
detail: err instanceof Error ? err.message : String(err),
|
|
348
|
-
|
|
375
|
+
// A thrown capture cannot establish that every paid receipt arrived.
|
|
376
|
+
usage: { ...sumSandboxUsage(events), tokensKnown: false, usdKnown: false },
|
|
377
|
+
events,
|
|
349
378
|
}
|
|
350
379
|
} finally {
|
|
351
380
|
if (timer) clearTimeout(timer)
|
|
@@ -451,6 +480,8 @@ async function loopedShot(
|
|
|
451
480
|
return {
|
|
452
481
|
artifact: completed.at(-1)?.artifact ?? '',
|
|
453
482
|
ok: false,
|
|
483
|
+
execution: pendingShot ? { phase: 'unknown', terminalOutcome: 'unknown' } : completed.at(-1)?.execution,
|
|
484
|
+
artifactAvailable: false,
|
|
454
485
|
usage: combinedUsage(pendingShot ? [...completed, { artifact: '', ok: false }] : completed),
|
|
455
486
|
events: completed.flatMap((shot) => shot.events ?? []),
|
|
456
487
|
detail: err instanceof Error ? err.message : String(err),
|
|
@@ -458,8 +489,10 @@ async function loopedShot(
|
|
|
458
489
|
}
|
|
459
490
|
|
|
460
491
|
const best = result.rounds.reduce((winner, candidate) => {
|
|
461
|
-
|
|
462
|
-
|
|
492
|
+
const candidateShot = shots.get(candidate.round)
|
|
493
|
+
const winnerShot = shots.get(winner.round)
|
|
494
|
+
const rank = (shot: BenchShotResult | undefined) => shot?.ok ? 2 : shot?.artifactAvailable ? 1 : 0
|
|
495
|
+
if (rank(candidateShot) !== rank(winnerShot)) return rank(candidateShot) > rank(winnerShot) ? candidate : winner
|
|
463
496
|
const a = scores.get(winner.round)
|
|
464
497
|
const b = scores.get(candidate.round)
|
|
465
498
|
if (!a) return candidate
|
|
@@ -472,6 +505,8 @@ async function loopedShot(
|
|
|
472
505
|
return {
|
|
473
506
|
artifact: best.artifact,
|
|
474
507
|
ok: shots.get(best.round)?.ok === true && best.artifact.trim().length > 0,
|
|
508
|
+
execution: shots.get(best.round)?.execution,
|
|
509
|
+
artifactAvailable: shots.get(best.round)?.artifactAvailable,
|
|
475
510
|
usage: combinedUsage([...shots.values()]),
|
|
476
511
|
events: [...shots.values()].flatMap((shot) => shot.events ?? []),
|
|
477
512
|
detail: JSON.stringify({
|
|
@@ -588,6 +623,7 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
|
|
|
588
623
|
const startedAt = Date.now()
|
|
589
624
|
let result: BenchCellTaskResult
|
|
590
625
|
let out: BenchShotResult | undefined
|
|
626
|
+
let invoked = false
|
|
591
627
|
try {
|
|
592
628
|
opts.signal?.throwIfAborted()
|
|
593
629
|
const shotInput = {
|
|
@@ -604,6 +640,7 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
|
|
|
604
640
|
...(opts.signal ? { signal: opts.signal } : {}),
|
|
605
641
|
...(opts.resolveClient ? { resolveClient: opts.resolveClient } : {}),
|
|
606
642
|
}
|
|
643
|
+
invoked = true
|
|
607
644
|
out = loopAttempts > 1 ? await loopedShot(shotInput, shot, loopAttempts) : await shot(shotInput)
|
|
608
645
|
const score: BenchScore = await job.adapter.judge(job.task, out.artifact)
|
|
609
646
|
result = {
|
|
@@ -614,6 +651,8 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
|
|
|
614
651
|
resolved: out.ok && score.resolved,
|
|
615
652
|
score: out.ok ? score.score : 0,
|
|
616
653
|
ok: out.ok,
|
|
654
|
+
execution: out.execution ?? { phase: out.ok ? 'started' : 'unknown', terminalOutcome: out.ok ? 'succeeded' : 'unknown' },
|
|
655
|
+
measurement: (out.artifactAvailable ?? out.ok) ? 'available' : 'unavailable',
|
|
617
656
|
...(out.detail ?? score.detail ? { detail: combineDetails(out.detail, score.detail) } : {}),
|
|
618
657
|
wallMs: Date.now() - startedAt,
|
|
619
658
|
artifact: out.artifact,
|
|
@@ -621,8 +660,7 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
|
|
|
621
660
|
...(out.events === undefined ? {} : { events: out.events }),
|
|
622
661
|
}
|
|
623
662
|
} catch (err) {
|
|
624
|
-
//
|
|
625
|
-
// resolve denominator (never a silent 0% that hides a harness outage).
|
|
663
|
+
// Missing results do not prove that dispatch or paid inference never occurred.
|
|
626
664
|
result = {
|
|
627
665
|
benchmark: job.benchmark,
|
|
628
666
|
cell: job.cell.label,
|
|
@@ -631,6 +669,11 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
|
|
|
631
669
|
resolved: false,
|
|
632
670
|
score: 0,
|
|
633
671
|
ok: false,
|
|
672
|
+
execution: out?.execution ?? {
|
|
673
|
+
phase: !invoked ? 'not-started' : out?.ok ? 'started' : 'unknown',
|
|
674
|
+
terminalOutcome: out?.ok ? 'succeeded' : 'unknown',
|
|
675
|
+
},
|
|
676
|
+
measurement: 'unavailable',
|
|
634
677
|
detail: err instanceof Error ? err.message.slice(0, 200) : String(err),
|
|
635
678
|
wallMs: Date.now() - startedAt,
|
|
636
679
|
...(out === undefined ? {} : { artifact: out.artifact }),
|
|
@@ -660,7 +703,7 @@ function aggregate(perTask: readonly BenchCellTaskResult[]): BenchLeaderboardRow
|
|
|
660
703
|
const key = `${r.benchmark}\u0000${r.cell}`
|
|
661
704
|
const e = byKey.get(key) ?? { benchmark: r.benchmark, cell: r.cell, n: 0, resolved: 0, errored: 0, scoreSum: 0 }
|
|
662
705
|
e.n += 1
|
|
663
|
-
if (
|
|
706
|
+
if ((r.measurement ?? (r.ok ? 'available' : 'unavailable')) === 'unavailable') e.errored += 1
|
|
664
707
|
else {
|
|
665
708
|
if (r.resolved) e.resolved += 1
|
|
666
709
|
e.scoreSum += r.score
|