@tangle-network/agent-bench 0.9.3 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-bench",
3
- "version": "0.9.3",
3
+ "version": "0.10.0",
4
4
  "type": "module",
5
5
  "description": "Benchmark adapters and execution for agent-runtime across coding, tool-use, RAG, memory, browser, and terminal tasks.",
6
6
  "repository": {
@@ -25,11 +25,11 @@
25
25
  }
26
26
  },
27
27
  "dependencies": {
28
- "@tangle-network/agent-eval": ">=0.176.0 <0.177.0",
29
- "@tangle-network/agent-interface": "^2.5.0",
30
- "@tangle-network/agent-knowledge": "^15.0.0",
28
+ "@tangle-network/agent-eval": ">=0.178.0 <0.179.0",
29
+ "@tangle-network/agent-interface": "^2.6.0",
30
+ "@tangle-network/agent-knowledge": "^15.0.1",
31
31
  "@tangle-network/sandbox": ">=0.36.4 <0.38.0",
32
- "@tangle-network/agent-runtime": "^0.201.0"
32
+ "@tangle-network/agent-runtime": "^0.203.0"
33
33
  },
34
34
  "devDependencies": {
35
35
  "@arethetypeswrong/cli": "0.18.5",
@@ -1,5 +1,6 @@
1
1
  import { execFile } from 'node:child_process'
2
- import { access, readdir, readFile } from 'node:fs/promises'
2
+ import { access, readdir, readFile, realpath } from 'node:fs/promises'
3
+ import { tmpdir } from 'node:os'
3
4
  import path from 'node:path'
4
5
  import { fileURLToPath } from 'node:url'
5
6
  import { promisify } from 'node:util'
@@ -8,6 +9,21 @@ const execFileAsync = promisify(execFile)
8
9
  const benchDir = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..')
9
10
  const sourceDir = path.join(benchDir, 'src')
10
11
 
12
+ export function packageNodeTestArgs(files, env = process.env) {
13
+ const raw = env.AGENT_BENCH_PACKAGE_TEST_CONCURRENCY
14
+ const concurrency = raw === undefined ? undefined : Number(raw)
15
+ if (raw !== undefined && (!/^[1-9]\d*$/.test(raw) || !Number.isSafeInteger(concurrency))) {
16
+ throw new Error('AGENT_BENCH_PACKAGE_TEST_CONCURRENCY must be a positive safe integer')
17
+ }
18
+ return [
19
+ '--test',
20
+ ...(concurrency === undefined ? [] : [`--test-concurrency=${concurrency}`]),
21
+ '--import',
22
+ 'tsx',
23
+ ...files,
24
+ ]
25
+ }
26
+
11
27
  export function resolvePackageTestTimeoutMs(env = process.env) {
12
28
  const raw = env.AGENT_BENCH_PACKAGE_TEST_TIMEOUT_MS
13
29
  if (raw === undefined) return undefined
@@ -49,6 +65,15 @@ export async function run(command, args, env = process.env) {
49
65
  }
50
66
  }
51
67
 
68
+ export async function runPythonTests(python, env = process.env) {
69
+ // Fixture roots must be physical paths; the production boundary rejects symlinked ancestors.
70
+ const physicalTemp = await realpath(env.TMPDIR || tmpdir())
71
+ await run(python, ['-m', 'unittest', 'discover', '-s', 'pier_agents', '-p', '*_test.py'], {
72
+ ...env,
73
+ TMPDIR: physicalTemp,
74
+ })
75
+ }
76
+
52
77
  async function main() {
53
78
  const python = path.join(benchDir, '.venv', 'bin', 'python')
54
79
  try {
@@ -74,7 +99,7 @@ async function main() {
74
99
  if (nodeTests.length > 0) {
75
100
  await run(
76
101
  process.execPath,
77
- ['--test', '--import', 'tsx', ...nodeTests.map((file) => path.relative(benchDir, file))],
102
+ packageNodeTestArgs(nodeTests.map((file) => path.relative(benchDir, file))),
78
103
  {
79
104
  ...process.env,
80
105
  TSX_TSCONFIG_PATH: 'tsconfig.public.json',
@@ -86,7 +111,7 @@ async function main() {
86
111
  await run('npx', ['vitest', 'run', ...vitestTests.map((file) => path.relative(benchDir, file))])
87
112
  }
88
113
 
89
- await run(python, ['-m', 'unittest', 'discover', '-s', 'pier_agents', '-p', '*_test.py'])
114
+ await runPythonTests(python)
90
115
 
91
116
  console.log(
92
117
  `package tests passed: ${tests.length}/${tests.length} TypeScript files (${nodeTests.length} node:test + ${vitestTests.length} vitest) + Pier bridge`,
@@ -1,6 +1,23 @@
1
1
  import assert from 'node:assert/strict'
2
2
  import { test } from 'node:test'
3
- import { resolvePackageTestTimeoutMs, run } from './run-package-tests.mjs'
3
+ import { mkdtemp, mkdir, readFile, realpath, rm, symlink, writeFile } from 'node:fs/promises'
4
+ import { tmpdir } from 'node:os'
5
+ import path from 'node:path'
6
+ import { packageNodeTestArgs, resolvePackageTestTimeoutMs, run, runPythonTests } from './run-package-tests.mjs'
7
+
8
+ test('package test concurrency reaches Node without changing selected files', () => {
9
+ const files = ['src/first.test.mts', 'src/second.test.ts']
10
+ assert.deepEqual(packageNodeTestArgs(files, {}), ['--test', '--import', 'tsx', ...files])
11
+ assert.deepEqual(packageNodeTestArgs(files, { AGENT_BENCH_PACKAGE_TEST_CONCURRENCY: '1' }), [
12
+ '--test', '--test-concurrency=1', '--import', 'tsx', ...files,
13
+ ])
14
+ for (const value of ['', '0', '-1', '1.5', 'Infinity', '9007199254740992']) {
15
+ assert.throws(
16
+ () => packageNodeTestArgs(files, { AGENT_BENCH_PACKAGE_TEST_CONCURRENCY: value }),
17
+ /must be a positive safe integer/,
18
+ )
19
+ }
20
+ })
4
21
 
5
22
  test('package test timeout is optional and caller-controlled', () => {
6
23
  assert.equal(resolvePackageTestTimeoutMs({}), undefined)
@@ -31,3 +48,25 @@ test('package test timeout reaches the child process', async () => {
31
48
  )
32
49
  assert.ok(Date.now() - startedAt < 2_000)
33
50
  })
51
+
52
+ test('Python fixtures use physical temporary paths without relaxing symlink guards', async () => {
53
+ const root = await mkdtemp(path.join(await realpath(tmpdir()), 'bench-python-temp-'))
54
+ try {
55
+ const target = path.join(root, 'physical')
56
+ const linked = path.join(root, 'linked')
57
+ const recorded = path.join(root, 'recorded.json')
58
+ const launcher = path.join(root, 'python-fixture')
59
+ await mkdir(target)
60
+ await symlink(target, linked, 'dir')
61
+ await writeFile(launcher, `#!${process.execPath}
62
+ const fs = require('node:fs'); fs.writeFileSync(process.env.RECORDED, JSON.stringify({ temporary: process.env.TMPDIR, args: process.argv.slice(2) }))
63
+ `, { mode: 0o755 })
64
+ await runPythonTests(launcher, { ...process.env, TMPDIR: linked, RECORDED: recorded })
65
+ assert.deepEqual(JSON.parse(await readFile(recorded, 'utf8')), {
66
+ temporary: target,
67
+ args: ['-m', 'unittest', 'discover', '-s', 'pier_agents', '-p', '*_test.py'],
68
+ })
69
+ } finally {
70
+ await rm(root, { recursive: true, force: true })
71
+ }
72
+ })
@@ -77,6 +77,22 @@ async function main(): Promise<void> {
77
77
  assert.equal(judgeFailure.perTask[0]?.ok, false)
78
78
  assert.equal(judgeFailure.perTask[0]?.artifact, 'PATCH', 'a judge outage retains the completed agent artifact')
79
79
  assert.equal(judgeFailure.perTask[0]?.usage?.costUsd, 0.04, 'a judge outage retains already incurred usage')
80
+ assert.equal(judgeFailure.perTask[0]?.measurement, 'unavailable')
81
+ assert.deepEqual(judgeFailure.perTask[0]?.execution, { phase: 'started', terminalOutcome: 'succeeded' })
82
+
83
+ const failedExecution = await runBenchmarks({
84
+ benchmarks: ['alpha'], cells: [{ label: 'paid-failure', model: 'm' }],
85
+ routerBaseUrl: 'x', routerKey: 'x', n: 1, resolveAdapter: resolveStub,
86
+ runShot: async () => ({
87
+ artifact: '', ok: false, artifactAvailable: true,
88
+ execution: { phase: 'started', terminalOutcome: 'failed' },
89
+ usage: { input: 23, output: 7, costUsd: 0.04 },
90
+ }),
91
+ })
92
+ assert.equal(failedExecution.perTask[0]?.measurement, 'available')
93
+ assert.equal(failedExecution.perTask[0]?.resolved, false)
94
+ assert.equal(failedExecution.rows[0]?.errored, 0, 'a measured paid failure stays in the comparison')
95
+ assert.equal(failedExecution.perTask[0]?.usage?.input, 23)
80
96
 
81
97
  const controller = new AbortController()
82
98
  let started = 0
@@ -93,6 +109,7 @@ async function main(): Promise<void> {
93
109
  })
94
110
  assert.equal(started, 1, 'cancellation prevents every queued model call')
95
111
  assert.equal(cancelled.perTask.length, 4, 'cancelled work remains visible')
112
+ assert.equal(cancelled.perTask.filter((row) => row.execution?.phase === 'not-started').length, 3)
96
113
  assert.equal(cancelled.perTask[0]?.usage?.costUsd, 0.01)
97
114
 
98
115
  const row = (b: string, c: string) => report.rows.find((r) => r.benchmark === b && r.cell === c)!
@@ -159,6 +176,20 @@ async function main(): Promise<void> {
159
176
  input: 11, output: 3, costUsd: 0.02, tokensKnown: false, usdKnown: false,
160
177
  }, 'an unreported retry preserves the measured floor without claiming complete accounting')
161
178
 
179
+ const measuredFailedRetry = await runBenchmarks({
180
+ benchmarks: ['alpha'], cells: [{ label: 'failed-retries', model: 'm' }],
181
+ routerBaseUrl: 'x', routerKey: 'x', resolveAdapter: resolveStub, n: 1, loopAttempts: 2,
182
+ runShot: async ({ attempt }) => ({
183
+ artifact: attempt === 1 ? '' : 'WRONG', ok: false, artifactAvailable: attempt === 2,
184
+ execution: { phase: 'started', terminalOutcome: 'failed' },
185
+ usage: { input: 11, output: 3, costUsd: 0.02 },
186
+ }),
187
+ })
188
+ assert.equal(measuredFailedRetry.perTask[0]?.artifact, 'WRONG')
189
+ assert.equal(measuredFailedRetry.perTask[0]?.measurement, 'available')
190
+ assert.equal(measuredFailedRetry.rows[0]?.errored, 0, 'a measured failed retry stays in the comparison')
191
+ assert.deepEqual(measuredFailedRetry.perTask[0]?.usage, { input: 22, output: 6, costUsd: 0.04 })
192
+
162
193
  let validAttempts = 0
163
194
  const validRetry = await runBenchmarks({
164
195
  benchmarks: ['alpha'], cells: [{ label: 'retrying', model: 'm' }],
@@ -212,6 +243,10 @@ async function main(): Promise<void> {
212
243
  const order: string[] = []
213
244
  let createdOptions: unknown
214
245
  let controlCredential: string | undefined
246
+ let completed = true
247
+ let text = 'fallback text'
248
+ let streamThrows = false
249
+ let setupFails = false
215
250
  const fakeClient = {
216
251
  async create(options: unknown) {
217
252
  createdOptions = options
@@ -219,13 +254,14 @@ async function main(): Promise<void> {
219
254
  id: 'box-default-shot',
220
255
  async exec(command: string, options?: { sessionId?: string }) {
221
256
  order.push(`exec:${command}:streams=${order.filter((x) => x.startsWith('stream:')).length}:session=${options?.sessionId ? 'yes' : 'no'}`)
222
- return { exitCode: 0, stdout: command === 'extract-patch' ? 'PATCH' : '', stderr: '' }
257
+ return { exitCode: setupFails && command === 'setup-repo' ? 1 : 0, stdout: command === 'extract-patch' ? 'PATCH' : '', stderr: '' }
223
258
  },
224
259
  async *streamPrompt(_prompt: string, options?: { sessionId?: string }) {
225
260
  order.push(`stream:session=${options?.sessionId ? 'yes' : 'no'}`)
226
261
  yield { type: 'llm_call', data: { tokensIn: 23, tokensOut: 7, costUsd: 0.04 } }
227
- yield { type: 'result', data: { finalText: 'fallback text' } }
228
- yield { type: 'done', data: { outcome: { type: 'completed' } } }
262
+ if (streamThrows) throw new Error('stream disconnected')
263
+ yield { type: 'result', data: { finalText: text, success: completed, status: completed ? 'success' : 'failed' } }
264
+ yield { type: 'done', data: { outcome: { type: completed ? 'completed' : 'failed' } } }
229
265
  },
230
266
  async delete() {
231
267
  order.push('delete')
@@ -260,6 +296,8 @@ async function main(): Promise<void> {
260
296
  assert.equal(boxy.rows[0]!.resolveRate, 1, 'boxExtract artifact is judged instead of fallback text')
261
297
  assert.equal(boxy.perTask[0]?.artifact, 'PATCH')
262
298
  assert.deepEqual(boxy.perTask[0]?.usage, { input: 23, output: 7, costUsd: 0.04 })
299
+ assert.deepEqual(boxy.perTask[0]?.execution, { phase: 'started', terminalOutcome: 'succeeded' })
300
+ assert.equal(boxy.perTask[0]?.measurement, 'available')
263
301
  assert.equal(controlCredential, 'sandbox-control-token', 'inference grant never authorizes sandbox control')
264
302
  assert.equal((createdOptions as { backend: { model: { apiKey: string } } }).backend.model.apiKey, 'model-grant-token')
265
303
  assert.deepEqual(
@@ -267,6 +305,49 @@ async function main(): Promise<void> {
267
305
  ['exec:setup-repo:streams=0:session=yes', 'stream:session=yes', 'exec:extract-patch:streams=1:session=yes'],
268
306
  'setup runs before the prompt stream, extract runs after the prompt stream, both in the same session',
269
307
  )
308
+ for (const failure of ['stream', 'parser'] as const) {
309
+ streamThrows = failure === 'stream'
310
+ const failed = await runBenchmarks({
311
+ benchmarks: ['boxy'], cells: [{ label: failure, model: 'm', backend: 'sandbox' }],
312
+ routerBaseUrl: 'x', routerKey: 'x',
313
+ resolveAdapter: () => ({ ...boxAdapter, output: { parse: () => { throw new Error('parser failed') } } }),
314
+ resolveClient: () => fakeClient as never,
315
+ })
316
+ assert.equal(failed.perTask[0]?.measurement, 'unavailable')
317
+ assert.equal(failed.perTask[0]?.execution?.phase, 'started')
318
+ assert.deepEqual(failed.perTask[0]?.usage, {
319
+ input: 23, output: 7, costUsd: 0.04, tokensKnown: false, usdKnown: false,
320
+ }, `${failure} failure retains paid receipts without claiming complete accounting`)
321
+ assert.equal(failed.perTask[0]?.events?.length, failure === 'stream' ? 1 : 3, 'observed events are retained once')
322
+ assert.match(failed.perTask[0]?.detail ?? '', failure === 'stream' ? /stream disconnected/ : /parser failed/)
323
+ }
324
+ streamThrows = false
325
+ setupFails = true
326
+ const setupFailure = await runBenchmarks({
327
+ benchmarks: ['boxy'], cells: [{ label: 'setup-failure', model: 'm', backend: 'sandbox' }],
328
+ routerBaseUrl: 'x', routerKey: 'x', resolveAdapter: () => boxAdapter,
329
+ resolveClient: () => fakeClient as never,
330
+ })
331
+ assert.deepEqual(setupFailure.perTask[0]?.execution, { phase: 'unknown', terminalOutcome: 'unknown' })
332
+ assert.equal(setupFailure.perTask[0]?.measurement, 'unavailable')
333
+ assert.equal(setupFailure.perTask[0]?.usage?.tokensKnown, false, 'no observed receipt does not prove zero usage')
334
+ assert.equal(setupFailure.perTask[0]?.usage?.usdKnown, false)
335
+ assert.equal(setupFailure.perTask[0]?.events?.length, 0)
336
+ setupFails = false
337
+ for (const success of [false, true]) {
338
+ completed = success
339
+ text = ''
340
+ const empty = await runBenchmarks({
341
+ benchmarks: ['alpha'], cells: [{ label: 'empty', model: 'm', backend: 'sandbox' }],
342
+ routerBaseUrl: 'x', routerKey: 'x', n: 1, resolveAdapter: resolveStub,
343
+ resolveClient: () => fakeClient as never,
344
+ })
345
+ assert.equal(empty.perTask[0]?.ok, false)
346
+ assert.equal(empty.perTask[0]?.measurement, 'available', 'captured empty output is a measurable failure')
347
+ assert.equal(empty.perTask[0]?.execution?.terminalOutcome, success ? 'succeeded' : 'failed')
348
+ assert.equal(empty.rows[0]?.errored, 0)
349
+ assert.equal(empty.perTask[0]?.usage?.input, 23)
350
+ }
270
351
  }
271
352
 
272
353
  // An unavailable benchmark (preflight throws) is skipped, not fatal; the sweep still runs the rest.
@@ -29,6 +29,7 @@
29
29
  */
30
30
 
31
31
  import { mkdirSync, writeFileSync } from 'node:fs'
32
+ import type { RunTerminalOutcome } from '@tangle-network/agent-eval'
32
33
  import type {
33
34
  AgentProfile,
34
35
  AgentRunSpec,
@@ -68,6 +69,13 @@ export interface BenchShotResult {
68
69
  /** Provider observations, including explicit unknown counters. Omitted when the shot reports none. */
69
70
  readonly usage?: ReturnType<typeof sumSandboxUsage>
70
71
  readonly events?: readonly SandboxEvent[]
72
+ /** Observed dispatch and terminal state, independent of artifact quality. */
73
+ readonly execution?: {
74
+ readonly phase: 'not-started' | 'started' | 'unknown'
75
+ readonly terminalOutcome: RunTerminalOutcome
76
+ }
77
+ /** Whether the artifact was captured without a read or extraction failure. */
78
+ readonly artifactAvailable?: boolean
71
79
  }
72
80
 
73
81
  /** Runs one (adapter, task, cell) shot. Defaults to `openSandboxRun`. */
@@ -137,9 +145,11 @@ export interface BenchCellTaskResult {
137
145
  readonly rep: number
138
146
  readonly resolved: boolean
139
147
  readonly score: number
140
- /** false = the shot threw or produced no artifact (infra/empty), excluded from the resolve
141
- * denominator so a harness outage can't masquerade as a 0% capability result. */
148
+ /** Whether execution completed successfully and produced a readable, nonempty artifact. */
142
149
  readonly ok: boolean
150
+ readonly execution?: BenchShotResult['execution']
151
+ /** Available failed attempts remain in comparisons; unavailable measurement is reported separately. */
152
+ readonly measurement?: 'available' | 'unavailable'
143
153
  readonly detail?: string
144
154
  readonly wallMs: number
145
155
  /** Exact bytes given to the benchmark judge, retained even when judging fails. */
@@ -237,11 +247,13 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
237
247
  }
238
248
  const controller = new AbortController()
239
249
  const timer = timeoutMs ? setTimeout(() => controller.abort(), timeoutMs) : undefined
250
+ const observedEvents: SandboxEvent[] = []
240
251
  const runOptions: OpenSandboxRunOptions = {
241
252
  agentRun,
242
253
  signal: signal ? AbortSignal.any([controller.signal, signal]) : controller.signal,
243
254
  runId: `bench:${adapter.name}:${task.id}:${uniq}`,
244
255
  scenarioId: task.id,
256
+ onSandboxEvent: (event) => { observedEvents.push(event) },
245
257
  }
246
258
  const boxSetup = adapter.boxSetup
247
259
  if (boxSetup) {
@@ -262,8 +274,15 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
262
274
  let result: BenchShotResult = { artifact: '', ok: false }
263
275
  try {
264
276
  run = await openSandboxRun(client, runOptions, deliverable)
277
+ result = { ...result, execution: { phase: 'unknown', terminalOutcome: 'unknown' } }
265
278
  const turn = await run.start(prompt ?? task.prompt)
266
- result = { artifact: '', ok: false, usage: sumSandboxUsage(turn.events), events: turn.events }
279
+ result = {
280
+ artifact: '', ok: false, usage: sumSandboxUsage(turn.events), events: turn.events,
281
+ execution: {
282
+ phase: 'started',
283
+ terminalOutcome: turn.outcome.success ? 'succeeded' : turn.outcome.status === 'failed' ? 'failed' : 'incomplete',
284
+ },
285
+ }
267
286
  // Event-stream deliverable (adapter.output ?? finalText) — the FALLBACK.
268
287
  let artifact = (turn.out ?? '').trim()
269
288
  let boxExtractError: string | undefined
@@ -336,16 +355,26 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
336
355
  result = {
337
356
  artifact,
338
357
  ok: turn.outcome.success && artifact.length > 0 && turn.readError === undefined && boxExtractError === undefined,
358
+ execution: result.execution,
359
+ artifactAvailable: turn.readError === undefined && boxExtractError === undefined,
339
360
  usage: result.usage,
340
361
  events: turn.events,
341
362
  ...(detail ? { detail } : {}),
342
363
  }
343
364
  } catch (err) {
365
+ const events = err instanceof SandboxRunAbortError ? err.events : observedEvents
344
366
  result = {
345
367
  ...result,
346
368
  ok: false,
369
+ artifactAvailable: false,
370
+ execution: {
371
+ phase: events.length > 0 ? 'started' : 'unknown',
372
+ terminalOutcome: result.execution?.terminalOutcome ?? 'unknown',
373
+ },
347
374
  detail: err instanceof Error ? err.message : String(err),
348
- ...(err instanceof SandboxRunAbortError ? { usage: sumSandboxUsage(err.events), events: err.events } : {}),
375
+ // A thrown capture cannot establish that every paid receipt arrived.
376
+ usage: { ...sumSandboxUsage(events), tokensKnown: false, usdKnown: false },
377
+ events,
349
378
  }
350
379
  } finally {
351
380
  if (timer) clearTimeout(timer)
@@ -451,6 +480,8 @@ async function loopedShot(
451
480
  return {
452
481
  artifact: completed.at(-1)?.artifact ?? '',
453
482
  ok: false,
483
+ execution: pendingShot ? { phase: 'unknown', terminalOutcome: 'unknown' } : completed.at(-1)?.execution,
484
+ artifactAvailable: false,
454
485
  usage: combinedUsage(pendingShot ? [...completed, { artifact: '', ok: false }] : completed),
455
486
  events: completed.flatMap((shot) => shot.events ?? []),
456
487
  detail: err instanceof Error ? err.message : String(err),
@@ -458,8 +489,10 @@ async function loopedShot(
458
489
  }
459
490
 
460
491
  const best = result.rounds.reduce((winner, candidate) => {
461
- if (shots.get(candidate.round)?.ok !== true) return winner
462
- if (shots.get(winner.round)?.ok !== true) return candidate
492
+ const candidateShot = shots.get(candidate.round)
493
+ const winnerShot = shots.get(winner.round)
494
+ const rank = (shot: BenchShotResult | undefined) => shot?.ok ? 2 : shot?.artifactAvailable ? 1 : 0
495
+ if (rank(candidateShot) !== rank(winnerShot)) return rank(candidateShot) > rank(winnerShot) ? candidate : winner
463
496
  const a = scores.get(winner.round)
464
497
  const b = scores.get(candidate.round)
465
498
  if (!a) return candidate
@@ -472,6 +505,8 @@ async function loopedShot(
472
505
  return {
473
506
  artifact: best.artifact,
474
507
  ok: shots.get(best.round)?.ok === true && best.artifact.trim().length > 0,
508
+ execution: shots.get(best.round)?.execution,
509
+ artifactAvailable: shots.get(best.round)?.artifactAvailable,
475
510
  usage: combinedUsage([...shots.values()]),
476
511
  events: [...shots.values()].flatMap((shot) => shot.events ?? []),
477
512
  detail: JSON.stringify({
@@ -588,6 +623,7 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
588
623
  const startedAt = Date.now()
589
624
  let result: BenchCellTaskResult
590
625
  let out: BenchShotResult | undefined
626
+ let invoked = false
591
627
  try {
592
628
  opts.signal?.throwIfAborted()
593
629
  const shotInput = {
@@ -604,6 +640,7 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
604
640
  ...(opts.signal ? { signal: opts.signal } : {}),
605
641
  ...(opts.resolveClient ? { resolveClient: opts.resolveClient } : {}),
606
642
  }
643
+ invoked = true
607
644
  out = loopAttempts > 1 ? await loopedShot(shotInput, shot, loopAttempts) : await shot(shotInput)
608
645
  const score: BenchScore = await job.adapter.judge(job.task, out.artifact)
609
646
  result = {
@@ -614,6 +651,8 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
614
651
  resolved: out.ok && score.resolved,
615
652
  score: out.ok ? score.score : 0,
616
653
  ok: out.ok,
654
+ execution: out.execution ?? { phase: out.ok ? 'started' : 'unknown', terminalOutcome: out.ok ? 'succeeded' : 'unknown' },
655
+ measurement: (out.artifactAvailable ?? out.ok) ? 'available' : 'unavailable',
617
656
  ...(out.detail ?? score.detail ? { detail: combineDetails(out.detail, score.detail) } : {}),
618
657
  wallMs: Date.now() - startedAt,
619
658
  artifact: out.artifact,
@@ -621,8 +660,7 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
621
660
  ...(out.events === undefined ? {} : { events: out.events }),
622
661
  }
623
662
  } catch (err) {
624
- // A thrown shot/judge is infra error for THIS cell-task: ok=false excludes it from the
625
- // resolve denominator (never a silent 0% that hides a harness outage).
663
+ // Missing results do not prove that dispatch or paid inference never occurred.
626
664
  result = {
627
665
  benchmark: job.benchmark,
628
666
  cell: job.cell.label,
@@ -631,6 +669,11 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
631
669
  resolved: false,
632
670
  score: 0,
633
671
  ok: false,
672
+ execution: out?.execution ?? {
673
+ phase: !invoked ? 'not-started' : out?.ok ? 'started' : 'unknown',
674
+ terminalOutcome: out?.ok ? 'succeeded' : 'unknown',
675
+ },
676
+ measurement: 'unavailable',
634
677
  detail: err instanceof Error ? err.message.slice(0, 200) : String(err),
635
678
  wallMs: Date.now() - startedAt,
636
679
  ...(out === undefined ? {} : { artifact: out.artifact }),
@@ -660,7 +703,7 @@ function aggregate(perTask: readonly BenchCellTaskResult[]): BenchLeaderboardRow
660
703
  const key = `${r.benchmark}\u0000${r.cell}`
661
704
  const e = byKey.get(key) ?? { benchmark: r.benchmark, cell: r.cell, n: 0, resolved: 0, errored: 0, scoreSum: 0 }
662
705
  e.n += 1
663
- if (!r.ok) e.errored += 1
706
+ if ((r.measurement ?? (r.ok ? 'available' : 'unavailable')) === 'unavailable') e.errored += 1
664
707
  else {
665
708
  if (r.resolved) e.resolved += 1
666
709
  e.scoreSum += r.score