@tangle-network/agent-bench 0.9.4 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +6 -0
- package/HARNESS.md +53 -15
- package/README.md +22 -0
- package/dist/index.d.ts +45 -5
- package/dist/index.js +180 -16
- package/dist/index.js.map +1 -1
- package/package.json +5 -5
- package/scripts/run-package-tests.mjs +12 -2
- package/scripts/run-package-tests.test.mjs +26 -1
- package/scripts/verify-packed-consumer.mjs +47 -1
- package/src/index.ts +3 -0
- package/src/run-benchmarks.test.mts +262 -4
- package/src/run-benchmarks.ts +195 -12
package/src/run-benchmarks.ts
CHANGED
|
@@ -29,11 +29,14 @@
|
|
|
29
29
|
*/
|
|
30
30
|
|
|
31
31
|
import { mkdirSync, writeFileSync } from 'node:fs'
|
|
32
|
+
import type { RunTerminalOutcome } from '@tangle-network/agent-eval'
|
|
32
33
|
import type {
|
|
33
34
|
AgentProfile,
|
|
34
35
|
AgentRunSpec,
|
|
35
36
|
Deliverable,
|
|
36
37
|
OpenSandboxRunOptions,
|
|
38
|
+
SandboxRun,
|
|
39
|
+
TurnResult,
|
|
37
40
|
} from '@tangle-network/agent-runtime/kernel'
|
|
38
41
|
import { openSandboxRun, SandboxRunAbortError, sumSandboxUsage } from '@tangle-network/agent-runtime/kernel'
|
|
39
42
|
import type { SandboxEvent } from '@tangle-network/sandbox'
|
|
@@ -60,6 +63,34 @@ export interface BenchCell {
|
|
|
60
63
|
readonly profile?: AgentProfile
|
|
61
64
|
}
|
|
62
65
|
|
|
66
|
+
/** Caller-owned work inside a managed benchmark shot. Final grading stays outside this callback. */
|
|
67
|
+
export interface BenchExecutionContext {
|
|
68
|
+
readonly prompt: string
|
|
69
|
+
readonly profile: AgentProfile
|
|
70
|
+
readonly benchmark: string
|
|
71
|
+
readonly taskId: string
|
|
72
|
+
readonly attempt: number
|
|
73
|
+
readonly signal: AbortSignal
|
|
74
|
+
/** Runtime owns session continuity. Bench owns capture, extraction, and close. */
|
|
75
|
+
readonly run: Pick<SandboxRun<string>, 'start' | 'resume' | 'box' | 'sessionId'>
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
export type BenchExecution = (context: BenchExecutionContext) => Promise<void>
|
|
79
|
+
|
|
80
|
+
/** One submitted sandbox prompt, including partial evidence when its invocation throws. */
|
|
81
|
+
export interface BenchPromptResult {
|
|
82
|
+
readonly attempt: number
|
|
83
|
+
readonly index: number
|
|
84
|
+
readonly method: 'start' | 'resume'
|
|
85
|
+
readonly prompt: string
|
|
86
|
+
readonly sessionId?: string
|
|
87
|
+
readonly outcome?: TurnResult<string>['outcome']
|
|
88
|
+
readonly events: readonly SandboxEvent[]
|
|
89
|
+
readonly usage: ReturnType<typeof sumSandboxUsage>
|
|
90
|
+
readonly readError?: string
|
|
91
|
+
readonly error?: string
|
|
92
|
+
}
|
|
93
|
+
|
|
63
94
|
/** A worker's artifact and observed execution evidence, before external grading. */
|
|
64
95
|
export interface BenchShotResult {
|
|
65
96
|
readonly artifact: string
|
|
@@ -68,6 +99,14 @@ export interface BenchShotResult {
|
|
|
68
99
|
/** Provider observations, including explicit unknown counters. Omitted when the shot reports none. */
|
|
69
100
|
readonly usage?: ReturnType<typeof sumSandboxUsage>
|
|
70
101
|
readonly events?: readonly SandboxEvent[]
|
|
102
|
+
readonly prompts?: readonly BenchPromptResult[]
|
|
103
|
+
/** Observed dispatch and terminal state, independent of artifact quality. */
|
|
104
|
+
readonly execution?: {
|
|
105
|
+
readonly phase: 'not-started' | 'started' | 'unknown'
|
|
106
|
+
readonly terminalOutcome: RunTerminalOutcome
|
|
107
|
+
}
|
|
108
|
+
/** Whether the artifact was captured without a read or extraction failure. */
|
|
109
|
+
readonly artifactAvailable?: boolean
|
|
71
110
|
}
|
|
72
111
|
|
|
73
112
|
/** Runs one (adapter, task, cell) shot. Defaults to `openSandboxRun`. */
|
|
@@ -79,6 +118,8 @@ export type BenchShot = (input: {
|
|
|
79
118
|
readonly prompt?: string
|
|
80
119
|
/** 1-based attempt index for looped runs. */
|
|
81
120
|
readonly attempt?: number
|
|
121
|
+
/** Custom shots own whether they consume this managed execution callback. */
|
|
122
|
+
readonly execute?: BenchExecution
|
|
82
123
|
readonly routerBaseUrl: string
|
|
83
124
|
readonly routerKey: string
|
|
84
125
|
/** Optional inference credential for the box; routerKey continues to authorize sandbox control. */
|
|
@@ -123,6 +164,8 @@ export interface RunBenchmarksOptions {
|
|
|
123
164
|
/** Self-verify each benchmark's judge against its gold artifact on the first task before spending
|
|
124
165
|
* model tokens; a benchmark whose judge rejects its own gold is recorded unavailable. Default true. */
|
|
125
166
|
readonly verifyJudge?: boolean
|
|
167
|
+
/** Caller policy inside each managed shot, including every refine-loop attempt. */
|
|
168
|
+
readonly execute?: BenchExecution
|
|
126
169
|
/** Test seam: a deterministic shot runner. Defaults to the `openSandboxRun` leaf. */
|
|
127
170
|
readonly runShot?: BenchShot
|
|
128
171
|
/** Test seam: resolve a benchmark key to an adapter. Defaults to the registry `resolveAdapter`. */
|
|
@@ -137,9 +180,11 @@ export interface BenchCellTaskResult {
|
|
|
137
180
|
readonly rep: number
|
|
138
181
|
readonly resolved: boolean
|
|
139
182
|
readonly score: number
|
|
140
|
-
/**
|
|
141
|
-
* denominator so a harness outage can't masquerade as a 0% capability result. */
|
|
183
|
+
/** Whether execution completed successfully and produced a readable, nonempty artifact. */
|
|
142
184
|
readonly ok: boolean
|
|
185
|
+
readonly execution?: BenchShotResult['execution']
|
|
186
|
+
/** Available failed attempts remain in comparisons; unavailable measurement is reported separately. */
|
|
187
|
+
readonly measurement?: 'available' | 'unavailable'
|
|
143
188
|
readonly detail?: string
|
|
144
189
|
readonly wallMs: number
|
|
145
190
|
/** Exact bytes given to the benchmark judge, retained even when judging fails. */
|
|
@@ -147,6 +192,7 @@ export interface BenchCellTaskResult {
|
|
|
147
192
|
readonly usage?: ReturnType<typeof sumSandboxUsage>
|
|
148
193
|
/** Worker events only; benchmark grading remains outside this trace. */
|
|
149
194
|
readonly events?: readonly SandboxEvent[]
|
|
195
|
+
readonly prompts?: readonly BenchPromptResult[]
|
|
150
196
|
}
|
|
151
197
|
|
|
152
198
|
export interface BenchLeaderboardRow {
|
|
@@ -194,7 +240,7 @@ function finalText(events: readonly SandboxEvent[]): string {
|
|
|
194
240
|
|
|
195
241
|
/** The default real-agent shot: one `openSandboxRun` over the cell's harness+model, deliverable
|
|
196
242
|
* extracted by the adapter's parser (or final text), abortable on `timeoutMs`. */
|
|
197
|
-
const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerBaseUrl, routerKey, modelApiKey, bridgeUrl, bridgeBearer, sandboxBaseUrl, timeoutMs, signal, resolveClient }) => {
|
|
243
|
+
const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, attempt = 1, execute, routerBaseUrl, routerKey, modelApiKey, bridgeUrl, bridgeBearer, sandboxBaseUrl, timeoutMs, signal, resolveClient }) => {
|
|
198
244
|
signal?.throwIfAborted()
|
|
199
245
|
const client = (resolveClient ?? resolveBenchClient)({
|
|
200
246
|
backend: cell.backend ?? 'router',
|
|
@@ -237,11 +283,13 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
|
|
|
237
283
|
}
|
|
238
284
|
const controller = new AbortController()
|
|
239
285
|
const timer = timeoutMs ? setTimeout(() => controller.abort(), timeoutMs) : undefined
|
|
286
|
+
const observedEvents: SandboxEvent[] = []
|
|
240
287
|
const runOptions: OpenSandboxRunOptions = {
|
|
241
288
|
agentRun,
|
|
242
289
|
signal: signal ? AbortSignal.any([controller.signal, signal]) : controller.signal,
|
|
243
290
|
runId: `bench:${adapter.name}:${task.id}:${uniq}`,
|
|
244
291
|
scenarioId: task.id,
|
|
292
|
+
onSandboxEvent: (event) => { observedEvents.push(event) },
|
|
245
293
|
}
|
|
246
294
|
const boxSetup = adapter.boxSetup
|
|
247
295
|
if (boxSetup) {
|
|
@@ -260,10 +308,96 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
|
|
|
260
308
|
}
|
|
261
309
|
let run: Awaited<ReturnType<typeof openSandboxRun<string>>> | undefined
|
|
262
310
|
let result: BenchShotResult = { artifact: '', ok: false }
|
|
311
|
+
const prompts: BenchPromptResult[] = []
|
|
312
|
+
const execution: {
|
|
313
|
+
lastTurn?: TurnResult<string>
|
|
314
|
+
active?: Promise<TurnResult<string>>
|
|
315
|
+
accepting: boolean
|
|
316
|
+
coordinationError?: Error
|
|
317
|
+
} = { accepting: true }
|
|
263
318
|
try {
|
|
264
319
|
run = await openSandboxRun(client, runOptions, deliverable)
|
|
265
|
-
|
|
266
|
-
|
|
320
|
+
result = { ...result, execution: { phase: 'unknown', terminalOutcome: 'unknown' } }
|
|
321
|
+
const managedRun = run
|
|
322
|
+
const invoke = (method: 'start' | 'resume', input: string): Promise<TurnResult<string>> => {
|
|
323
|
+
if (!execution.accepting || execution.active) {
|
|
324
|
+
execution.coordinationError = new Error(!execution.accepting
|
|
325
|
+
? 'benchmark execution callback has settled'
|
|
326
|
+
: 'benchmark execution requires sequential start/resume calls')
|
|
327
|
+
throw execution.coordinationError
|
|
328
|
+
}
|
|
329
|
+
runOptions.signal.throwIfAborted()
|
|
330
|
+
const index = prompts.length
|
|
331
|
+
const eventStart = observedEvents.length
|
|
332
|
+
execution.lastTurn = undefined
|
|
333
|
+
const capture = async (): Promise<TurnResult<string>> => {
|
|
334
|
+
let turn: TurnResult<string> | undefined
|
|
335
|
+
let failure: unknown
|
|
336
|
+
try {
|
|
337
|
+
turn = await managedRun[method](input)
|
|
338
|
+
return turn
|
|
339
|
+
} catch (error) {
|
|
340
|
+
failure = error
|
|
341
|
+
throw error
|
|
342
|
+
} finally {
|
|
343
|
+
// The observer captures generic stream/parser failures as well as aborts.
|
|
344
|
+
const events = observedEvents.slice(eventStart)
|
|
345
|
+
if (turn) execution.lastTurn = { ...turn, events, outcome: { ...turn.outcome } }
|
|
346
|
+
let sessionId: string | undefined
|
|
347
|
+
try { sessionId = managedRun.sessionId } catch { /* Creation may have failed. */ }
|
|
348
|
+
const readError = turn?.readError ?? (failure instanceof SandboxRunAbortError ? failure.readError : undefined)
|
|
349
|
+
prompts.push({
|
|
350
|
+
attempt, index, method, prompt: input,
|
|
351
|
+
...(sessionId === undefined ? {} : { sessionId }),
|
|
352
|
+
...(turn === undefined ? {} : { outcome: { ...turn.outcome } }),
|
|
353
|
+
events,
|
|
354
|
+
usage: turn === undefined
|
|
355
|
+
? { ...sumSandboxUsage(events), tokensKnown: false, usdKnown: false }
|
|
356
|
+
: sumSandboxUsage(events),
|
|
357
|
+
...(readError === undefined ? {} : { readError }),
|
|
358
|
+
...(turn !== undefined ? {} : { error: failure instanceof Error ? failure.message : String(failure) }),
|
|
359
|
+
})
|
|
360
|
+
}
|
|
361
|
+
}
|
|
362
|
+
const pending = capture()
|
|
363
|
+
execution.active = pending
|
|
364
|
+
// Observe even a caller's unawaited invocation; cleanup must await its capture.
|
|
365
|
+
void pending.then(() => { execution.active = undefined }, () => { execution.active = undefined })
|
|
366
|
+
return pending
|
|
367
|
+
}
|
|
368
|
+
const context: BenchExecutionContext = {
|
|
369
|
+
prompt: prompt ?? task.prompt, profile: structuredClone(profile), benchmark: adapter.name, taskId: task.id, attempt,
|
|
370
|
+
signal: runOptions.signal,
|
|
371
|
+
run: {
|
|
372
|
+
start: (input) => invoke('start', input),
|
|
373
|
+
resume: (input) => invoke('resume', input),
|
|
374
|
+
get box() { return managedRun.box },
|
|
375
|
+
get sessionId() { return managedRun.sessionId },
|
|
376
|
+
},
|
|
377
|
+
}
|
|
378
|
+
try {
|
|
379
|
+
await executeWithSignal(async () => {
|
|
380
|
+
if (execute) await execute(context)
|
|
381
|
+
else await context.run.start(context.prompt)
|
|
382
|
+
}, runOptions.signal)
|
|
383
|
+
} catch (error) {
|
|
384
|
+
if (execution.active) controller.abort()
|
|
385
|
+
throw error
|
|
386
|
+
} finally {
|
|
387
|
+
execution.accepting = false
|
|
388
|
+
await execution.active?.catch(() => undefined)
|
|
389
|
+
}
|
|
390
|
+
if (execution.coordinationError) throw execution.coordinationError
|
|
391
|
+
runOptions.signal.throwIfAborted()
|
|
392
|
+
const turn = execution.lastTurn
|
|
393
|
+
if (!turn) throw new Error(prompts.at(-1)?.error ?? 'benchmark execution returned without a completed prompt')
|
|
394
|
+
result = {
|
|
395
|
+
artifact: '', ok: false, usage: promptUsage(prompts), events: observedEvents, prompts,
|
|
396
|
+
execution: {
|
|
397
|
+
phase: 'started',
|
|
398
|
+
terminalOutcome: turn.outcome.success ? 'succeeded' : turn.outcome.status === 'failed' ? 'failed' : 'incomplete',
|
|
399
|
+
},
|
|
400
|
+
}
|
|
267
401
|
// Event-stream deliverable (adapter.output ?? finalText) — the FALLBACK.
|
|
268
402
|
let artifact = (turn.out ?? '').trim()
|
|
269
403
|
let boxExtractError: string | undefined
|
|
@@ -336,16 +470,27 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
|
|
|
336
470
|
result = {
|
|
337
471
|
artifact,
|
|
338
472
|
ok: turn.outcome.success && artifact.length > 0 && turn.readError === undefined && boxExtractError === undefined,
|
|
473
|
+
execution: result.execution,
|
|
474
|
+
artifactAvailable: turn.readError === undefined && boxExtractError === undefined,
|
|
339
475
|
usage: result.usage,
|
|
340
|
-
events:
|
|
476
|
+
events: observedEvents,
|
|
477
|
+
prompts,
|
|
341
478
|
...(detail ? { detail } : {}),
|
|
342
479
|
}
|
|
343
480
|
} catch (err) {
|
|
481
|
+
const events = observedEvents
|
|
344
482
|
result = {
|
|
345
483
|
...result,
|
|
346
484
|
ok: false,
|
|
485
|
+
artifactAvailable: false,
|
|
486
|
+
execution: {
|
|
487
|
+
phase: events.length > 0 ? 'started' : 'unknown',
|
|
488
|
+
terminalOutcome: result.execution?.terminalOutcome ?? 'unknown',
|
|
489
|
+
},
|
|
347
490
|
detail: err instanceof Error ? err.message : String(err),
|
|
348
|
-
|
|
491
|
+
usage: promptUsage(prompts),
|
|
492
|
+
events,
|
|
493
|
+
prompts,
|
|
349
494
|
}
|
|
350
495
|
} finally {
|
|
351
496
|
if (timer) clearTimeout(timer)
|
|
@@ -359,6 +504,25 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
|
|
|
359
504
|
return result
|
|
360
505
|
}
|
|
361
506
|
|
|
507
|
+
/** Stop waiting on policy work when cancelled; the caller must also cancel its external effects. */
|
|
508
|
+
async function executeWithSignal(execute: () => Promise<void>, signal: AbortSignal): Promise<void> {
|
|
509
|
+
signal.throwIfAborted()
|
|
510
|
+
let onAbort: (() => void) | undefined
|
|
511
|
+
const aborted = new Promise<never>((_resolve, reject) => {
|
|
512
|
+
onAbort = () => reject(signal.reason ?? new Error('aborted'))
|
|
513
|
+
signal.addEventListener('abort', onAbort, { once: true })
|
|
514
|
+
})
|
|
515
|
+
try {
|
|
516
|
+
await Promise.race([Promise.resolve().then(execute), aborted])
|
|
517
|
+
} finally {
|
|
518
|
+
if (onAbort) signal.removeEventListener('abort', onAbort)
|
|
519
|
+
}
|
|
520
|
+
}
|
|
521
|
+
|
|
522
|
+
function promptUsage(prompts: readonly BenchPromptResult[]): ReturnType<typeof sumSandboxUsage> {
|
|
523
|
+
return combinedUsage(prompts.map((prompt) => ({ artifact: '', ok: false, usage: prompt.usage })))
|
|
524
|
+
}
|
|
525
|
+
|
|
362
526
|
function parseMaybeJson(value: string): unknown {
|
|
363
527
|
try {
|
|
364
528
|
return JSON.parse(value) as unknown
|
|
@@ -451,15 +615,20 @@ async function loopedShot(
|
|
|
451
615
|
return {
|
|
452
616
|
artifact: completed.at(-1)?.artifact ?? '',
|
|
453
617
|
ok: false,
|
|
618
|
+
execution: pendingShot ? { phase: 'unknown', terminalOutcome: 'unknown' } : completed.at(-1)?.execution,
|
|
619
|
+
artifactAvailable: false,
|
|
454
620
|
usage: combinedUsage(pendingShot ? [...completed, { artifact: '', ok: false }] : completed),
|
|
455
621
|
events: completed.flatMap((shot) => shot.events ?? []),
|
|
622
|
+
prompts: completed.flatMap((shot) => shot.prompts ?? []),
|
|
456
623
|
detail: err instanceof Error ? err.message : String(err),
|
|
457
624
|
}
|
|
458
625
|
}
|
|
459
626
|
|
|
460
627
|
const best = result.rounds.reduce((winner, candidate) => {
|
|
461
|
-
|
|
462
|
-
|
|
628
|
+
const candidateShot = shots.get(candidate.round)
|
|
629
|
+
const winnerShot = shots.get(winner.round)
|
|
630
|
+
const rank = (shot: BenchShotResult | undefined) => shot?.ok ? 2 : shot?.artifactAvailable ? 1 : 0
|
|
631
|
+
if (rank(candidateShot) !== rank(winnerShot)) return rank(candidateShot) > rank(winnerShot) ? candidate : winner
|
|
463
632
|
const a = scores.get(winner.round)
|
|
464
633
|
const b = scores.get(candidate.round)
|
|
465
634
|
if (!a) return candidate
|
|
@@ -472,8 +641,11 @@ async function loopedShot(
|
|
|
472
641
|
return {
|
|
473
642
|
artifact: best.artifact,
|
|
474
643
|
ok: shots.get(best.round)?.ok === true && best.artifact.trim().length > 0,
|
|
644
|
+
execution: shots.get(best.round)?.execution,
|
|
645
|
+
artifactAvailable: shots.get(best.round)?.artifactAvailable,
|
|
475
646
|
usage: combinedUsage([...shots.values()]),
|
|
476
647
|
events: [...shots.values()].flatMap((shot) => shot.events ?? []),
|
|
648
|
+
prompts: [...shots.values()].flatMap((shot) => shot.prompts ?? []),
|
|
477
649
|
detail: JSON.stringify({
|
|
478
650
|
mode: 'refine-loop',
|
|
479
651
|
attempts: result.rounds.length,
|
|
@@ -588,6 +760,7 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
|
|
|
588
760
|
const startedAt = Date.now()
|
|
589
761
|
let result: BenchCellTaskResult
|
|
590
762
|
let out: BenchShotResult | undefined
|
|
763
|
+
let invoked = false
|
|
591
764
|
try {
|
|
592
765
|
opts.signal?.throwIfAborted()
|
|
593
766
|
const shotInput = {
|
|
@@ -603,7 +776,9 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
|
|
|
603
776
|
...(opts.timeoutMs ? { timeoutMs: opts.timeoutMs } : {}),
|
|
604
777
|
...(opts.signal ? { signal: opts.signal } : {}),
|
|
605
778
|
...(opts.resolveClient ? { resolveClient: opts.resolveClient } : {}),
|
|
779
|
+
...(opts.execute ? { execute: opts.execute } : {}),
|
|
606
780
|
}
|
|
781
|
+
invoked = true
|
|
607
782
|
out = loopAttempts > 1 ? await loopedShot(shotInput, shot, loopAttempts) : await shot(shotInput)
|
|
608
783
|
const score: BenchScore = await job.adapter.judge(job.task, out.artifact)
|
|
609
784
|
result = {
|
|
@@ -614,15 +789,17 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
|
|
|
614
789
|
resolved: out.ok && score.resolved,
|
|
615
790
|
score: out.ok ? score.score : 0,
|
|
616
791
|
ok: out.ok,
|
|
792
|
+
execution: out.execution ?? { phase: out.ok ? 'started' : 'unknown', terminalOutcome: out.ok ? 'succeeded' : 'unknown' },
|
|
793
|
+
measurement: (out.artifactAvailable ?? out.ok) ? 'available' : 'unavailable',
|
|
617
794
|
...(out.detail ?? score.detail ? { detail: combineDetails(out.detail, score.detail) } : {}),
|
|
618
795
|
wallMs: Date.now() - startedAt,
|
|
619
796
|
artifact: out.artifact,
|
|
620
797
|
...(out.usage === undefined ? {} : { usage: out.usage }),
|
|
621
798
|
...(out.events === undefined ? {} : { events: out.events }),
|
|
799
|
+
...(out.prompts === undefined ? {} : { prompts: out.prompts }),
|
|
622
800
|
}
|
|
623
801
|
} catch (err) {
|
|
624
|
-
//
|
|
625
|
-
// resolve denominator (never a silent 0% that hides a harness outage).
|
|
802
|
+
// Missing results do not prove that dispatch or paid inference never occurred.
|
|
626
803
|
result = {
|
|
627
804
|
benchmark: job.benchmark,
|
|
628
805
|
cell: job.cell.label,
|
|
@@ -631,11 +808,17 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
|
|
|
631
808
|
resolved: false,
|
|
632
809
|
score: 0,
|
|
633
810
|
ok: false,
|
|
811
|
+
execution: out?.execution ?? {
|
|
812
|
+
phase: !invoked ? 'not-started' : out?.ok ? 'started' : 'unknown',
|
|
813
|
+
terminalOutcome: out?.ok ? 'succeeded' : 'unknown',
|
|
814
|
+
},
|
|
815
|
+
measurement: 'unavailable',
|
|
634
816
|
detail: err instanceof Error ? err.message.slice(0, 200) : String(err),
|
|
635
817
|
wallMs: Date.now() - startedAt,
|
|
636
818
|
...(out === undefined ? {} : { artifact: out.artifact }),
|
|
637
819
|
...(out?.usage === undefined ? {} : { usage: out.usage }),
|
|
638
820
|
...(out?.events === undefined ? {} : { events: out.events }),
|
|
821
|
+
...(out?.prompts === undefined ? {} : { prompts: out.prompts }),
|
|
639
822
|
}
|
|
640
823
|
}
|
|
641
824
|
void index
|
|
@@ -660,7 +843,7 @@ function aggregate(perTask: readonly BenchCellTaskResult[]): BenchLeaderboardRow
|
|
|
660
843
|
const key = `${r.benchmark}\u0000${r.cell}`
|
|
661
844
|
const e = byKey.get(key) ?? { benchmark: r.benchmark, cell: r.cell, n: 0, resolved: 0, errored: 0, scoreSum: 0 }
|
|
662
845
|
e.n += 1
|
|
663
|
-
if (
|
|
846
|
+
if ((r.measurement ?? (r.ok ? 'available' : 'unavailable')) === 'unavailable') e.errored += 1
|
|
664
847
|
else {
|
|
665
848
|
if (r.resolved) e.resolved += 1
|
|
666
849
|
e.scoreSum += r.score
|