@tangle-network/agent-bench 0.9.4 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -29,11 +29,14 @@
29
29
  */
30
30
 
31
31
  import { mkdirSync, writeFileSync } from 'node:fs'
32
+ import type { RunTerminalOutcome } from '@tangle-network/agent-eval'
32
33
  import type {
33
34
  AgentProfile,
34
35
  AgentRunSpec,
35
36
  Deliverable,
36
37
  OpenSandboxRunOptions,
38
+ SandboxRun,
39
+ TurnResult,
37
40
  } from '@tangle-network/agent-runtime/kernel'
38
41
  import { openSandboxRun, SandboxRunAbortError, sumSandboxUsage } from '@tangle-network/agent-runtime/kernel'
39
42
  import type { SandboxEvent } from '@tangle-network/sandbox'
@@ -60,6 +63,34 @@ export interface BenchCell {
60
63
  readonly profile?: AgentProfile
61
64
  }
62
65
 
66
+ /** Caller-owned work inside a managed benchmark shot. Final grading stays outside this callback. */
67
+ export interface BenchExecutionContext {
68
+ readonly prompt: string
69
+ readonly profile: AgentProfile
70
+ readonly benchmark: string
71
+ readonly taskId: string
72
+ readonly attempt: number
73
+ readonly signal: AbortSignal
74
+ /** Runtime owns session continuity. Bench owns capture, extraction, and close. */
75
+ readonly run: Pick<SandboxRun<string>, 'start' | 'resume' | 'box' | 'sessionId'>
76
+ }
77
+
78
+ export type BenchExecution = (context: BenchExecutionContext) => Promise<void>
79
+
80
+ /** One submitted sandbox prompt, including partial evidence when its invocation throws. */
81
+ export interface BenchPromptResult {
82
+ readonly attempt: number
83
+ readonly index: number
84
+ readonly method: 'start' | 'resume'
85
+ readonly prompt: string
86
+ readonly sessionId?: string
87
+ readonly outcome?: TurnResult<string>['outcome']
88
+ readonly events: readonly SandboxEvent[]
89
+ readonly usage: ReturnType<typeof sumSandboxUsage>
90
+ readonly readError?: string
91
+ readonly error?: string
92
+ }
93
+
63
94
  /** A worker's artifact and observed execution evidence, before external grading. */
64
95
  export interface BenchShotResult {
65
96
  readonly artifact: string
@@ -68,6 +99,14 @@ export interface BenchShotResult {
68
99
  /** Provider observations, including explicit unknown counters. Omitted when the shot reports none. */
69
100
  readonly usage?: ReturnType<typeof sumSandboxUsage>
70
101
  readonly events?: readonly SandboxEvent[]
102
+ readonly prompts?: readonly BenchPromptResult[]
103
+ /** Observed dispatch and terminal state, independent of artifact quality. */
104
+ readonly execution?: {
105
+ readonly phase: 'not-started' | 'started' | 'unknown'
106
+ readonly terminalOutcome: RunTerminalOutcome
107
+ }
108
+ /** Whether the artifact was captured without a read or extraction failure. */
109
+ readonly artifactAvailable?: boolean
71
110
  }
72
111
 
73
112
  /** Runs one (adapter, task, cell) shot. Defaults to `openSandboxRun`. */
@@ -79,6 +118,8 @@ export type BenchShot = (input: {
79
118
  readonly prompt?: string
80
119
  /** 1-based attempt index for looped runs. */
81
120
  readonly attempt?: number
121
+ /** Custom shots own whether they consume this managed execution callback. */
122
+ readonly execute?: BenchExecution
82
123
  readonly routerBaseUrl: string
83
124
  readonly routerKey: string
84
125
  /** Optional inference credential for the box; routerKey continues to authorize sandbox control. */
@@ -123,6 +164,8 @@ export interface RunBenchmarksOptions {
123
164
  /** Self-verify each benchmark's judge against its gold artifact on the first task before spending
124
165
  * model tokens; a benchmark whose judge rejects its own gold is recorded unavailable. Default true. */
125
166
  readonly verifyJudge?: boolean
167
+ /** Caller policy inside each managed shot, including every refine-loop attempt. */
168
+ readonly execute?: BenchExecution
126
169
  /** Test seam: a deterministic shot runner. Defaults to the `openSandboxRun` leaf. */
127
170
  readonly runShot?: BenchShot
128
171
  /** Test seam: resolve a benchmark key to an adapter. Defaults to the registry `resolveAdapter`. */
@@ -137,9 +180,11 @@ export interface BenchCellTaskResult {
137
180
  readonly rep: number
138
181
  readonly resolved: boolean
139
182
  readonly score: number
140
- /** false = the shot threw or produced no artifact (infra/empty), excluded from the resolve
141
- * denominator so a harness outage can't masquerade as a 0% capability result. */
183
+ /** Whether execution completed successfully and produced a readable, nonempty artifact. */
142
184
  readonly ok: boolean
185
+ readonly execution?: BenchShotResult['execution']
186
+ /** Available failed attempts remain in comparisons; unavailable measurement is reported separately. */
187
+ readonly measurement?: 'available' | 'unavailable'
143
188
  readonly detail?: string
144
189
  readonly wallMs: number
145
190
  /** Exact bytes given to the benchmark judge, retained even when judging fails. */
@@ -147,6 +192,7 @@ export interface BenchCellTaskResult {
147
192
  readonly usage?: ReturnType<typeof sumSandboxUsage>
148
193
  /** Worker events only; benchmark grading remains outside this trace. */
149
194
  readonly events?: readonly SandboxEvent[]
195
+ readonly prompts?: readonly BenchPromptResult[]
150
196
  }
151
197
 
152
198
  export interface BenchLeaderboardRow {
@@ -194,7 +240,7 @@ function finalText(events: readonly SandboxEvent[]): string {
194
240
 
195
241
  /** The default real-agent shot: one `openSandboxRun` over the cell's harness+model, deliverable
196
242
  * extracted by the adapter's parser (or final text), abortable on `timeoutMs`. */
197
- const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerBaseUrl, routerKey, modelApiKey, bridgeUrl, bridgeBearer, sandboxBaseUrl, timeoutMs, signal, resolveClient }) => {
243
+ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, attempt = 1, execute, routerBaseUrl, routerKey, modelApiKey, bridgeUrl, bridgeBearer, sandboxBaseUrl, timeoutMs, signal, resolveClient }) => {
198
244
  signal?.throwIfAborted()
199
245
  const client = (resolveClient ?? resolveBenchClient)({
200
246
  backend: cell.backend ?? 'router',
@@ -237,11 +283,13 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
237
283
  }
238
284
  const controller = new AbortController()
239
285
  const timer = timeoutMs ? setTimeout(() => controller.abort(), timeoutMs) : undefined
286
+ const observedEvents: SandboxEvent[] = []
240
287
  const runOptions: OpenSandboxRunOptions = {
241
288
  agentRun,
242
289
  signal: signal ? AbortSignal.any([controller.signal, signal]) : controller.signal,
243
290
  runId: `bench:${adapter.name}:${task.id}:${uniq}`,
244
291
  scenarioId: task.id,
292
+ onSandboxEvent: (event) => { observedEvents.push(event) },
245
293
  }
246
294
  const boxSetup = adapter.boxSetup
247
295
  if (boxSetup) {
@@ -260,10 +308,96 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
260
308
  }
261
309
  let run: Awaited<ReturnType<typeof openSandboxRun<string>>> | undefined
262
310
  let result: BenchShotResult = { artifact: '', ok: false }
311
+ const prompts: BenchPromptResult[] = []
312
+ const execution: {
313
+ lastTurn?: TurnResult<string>
314
+ active?: Promise<TurnResult<string>>
315
+ accepting: boolean
316
+ coordinationError?: Error
317
+ } = { accepting: true }
263
318
  try {
264
319
  run = await openSandboxRun(client, runOptions, deliverable)
265
- const turn = await run.start(prompt ?? task.prompt)
266
- result = { artifact: '', ok: false, usage: sumSandboxUsage(turn.events), events: turn.events }
320
+ result = { ...result, execution: { phase: 'unknown', terminalOutcome: 'unknown' } }
321
+ const managedRun = run
322
+ const invoke = (method: 'start' | 'resume', input: string): Promise<TurnResult<string>> => {
323
+ if (!execution.accepting || execution.active) {
324
+ execution.coordinationError = new Error(!execution.accepting
325
+ ? 'benchmark execution callback has settled'
326
+ : 'benchmark execution requires sequential start/resume calls')
327
+ throw execution.coordinationError
328
+ }
329
+ runOptions.signal.throwIfAborted()
330
+ const index = prompts.length
331
+ const eventStart = observedEvents.length
332
+ execution.lastTurn = undefined
333
+ const capture = async (): Promise<TurnResult<string>> => {
334
+ let turn: TurnResult<string> | undefined
335
+ let failure: unknown
336
+ try {
337
+ turn = await managedRun[method](input)
338
+ return turn
339
+ } catch (error) {
340
+ failure = error
341
+ throw error
342
+ } finally {
343
+ // The observer captures generic stream/parser failures as well as aborts.
344
+ const events = observedEvents.slice(eventStart)
345
+ if (turn) execution.lastTurn = { ...turn, events, outcome: { ...turn.outcome } }
346
+ let sessionId: string | undefined
347
+ try { sessionId = managedRun.sessionId } catch { /* Creation may have failed. */ }
348
+ const readError = turn?.readError ?? (failure instanceof SandboxRunAbortError ? failure.readError : undefined)
349
+ prompts.push({
350
+ attempt, index, method, prompt: input,
351
+ ...(sessionId === undefined ? {} : { sessionId }),
352
+ ...(turn === undefined ? {} : { outcome: { ...turn.outcome } }),
353
+ events,
354
+ usage: turn === undefined
355
+ ? { ...sumSandboxUsage(events), tokensKnown: false, usdKnown: false }
356
+ : sumSandboxUsage(events),
357
+ ...(readError === undefined ? {} : { readError }),
358
+ ...(turn !== undefined ? {} : { error: failure instanceof Error ? failure.message : String(failure) }),
359
+ })
360
+ }
361
+ }
362
+ const pending = capture()
363
+ execution.active = pending
364
+ // Observe even a caller's unawaited invocation; cleanup must await its capture.
365
+ void pending.then(() => { execution.active = undefined }, () => { execution.active = undefined })
366
+ return pending
367
+ }
368
+ const context: BenchExecutionContext = {
369
+ prompt: prompt ?? task.prompt, profile: structuredClone(profile), benchmark: adapter.name, taskId: task.id, attempt,
370
+ signal: runOptions.signal,
371
+ run: {
372
+ start: (input) => invoke('start', input),
373
+ resume: (input) => invoke('resume', input),
374
+ get box() { return managedRun.box },
375
+ get sessionId() { return managedRun.sessionId },
376
+ },
377
+ }
378
+ try {
379
+ await executeWithSignal(async () => {
380
+ if (execute) await execute(context)
381
+ else await context.run.start(context.prompt)
382
+ }, runOptions.signal)
383
+ } catch (error) {
384
+ if (execution.active) controller.abort()
385
+ throw error
386
+ } finally {
387
+ execution.accepting = false
388
+ await execution.active?.catch(() => undefined)
389
+ }
390
+ if (execution.coordinationError) throw execution.coordinationError
391
+ runOptions.signal.throwIfAborted()
392
+ const turn = execution.lastTurn
393
+ if (!turn) throw new Error(prompts.at(-1)?.error ?? 'benchmark execution returned without a completed prompt')
394
+ result = {
395
+ artifact: '', ok: false, usage: promptUsage(prompts), events: observedEvents, prompts,
396
+ execution: {
397
+ phase: 'started',
398
+ terminalOutcome: turn.outcome.success ? 'succeeded' : turn.outcome.status === 'failed' ? 'failed' : 'incomplete',
399
+ },
400
+ }
267
401
  // Event-stream deliverable (adapter.output ?? finalText) — the FALLBACK.
268
402
  let artifact = (turn.out ?? '').trim()
269
403
  let boxExtractError: string | undefined
@@ -336,16 +470,27 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
336
470
  result = {
337
471
  artifact,
338
472
  ok: turn.outcome.success && artifact.length > 0 && turn.readError === undefined && boxExtractError === undefined,
473
+ execution: result.execution,
474
+ artifactAvailable: turn.readError === undefined && boxExtractError === undefined,
339
475
  usage: result.usage,
340
- events: turn.events,
476
+ events: observedEvents,
477
+ prompts,
341
478
  ...(detail ? { detail } : {}),
342
479
  }
343
480
  } catch (err) {
481
+ const events = observedEvents
344
482
  result = {
345
483
  ...result,
346
484
  ok: false,
485
+ artifactAvailable: false,
486
+ execution: {
487
+ phase: events.length > 0 ? 'started' : 'unknown',
488
+ terminalOutcome: result.execution?.terminalOutcome ?? 'unknown',
489
+ },
347
490
  detail: err instanceof Error ? err.message : String(err),
348
- ...(err instanceof SandboxRunAbortError ? { usage: sumSandboxUsage(err.events), events: err.events } : {}),
491
+ usage: promptUsage(prompts),
492
+ events,
493
+ prompts,
349
494
  }
350
495
  } finally {
351
496
  if (timer) clearTimeout(timer)
@@ -359,6 +504,25 @@ const openSandboxShot: BenchShot = async ({ adapter, task, cell, prompt, routerB
359
504
  return result
360
505
  }
361
506
 
507
+ /** Stop waiting on policy work when cancelled; the caller must also cancel its external effects. */
508
+ async function executeWithSignal(execute: () => Promise<void>, signal: AbortSignal): Promise<void> {
509
+ signal.throwIfAborted()
510
+ let onAbort: (() => void) | undefined
511
+ const aborted = new Promise<never>((_resolve, reject) => {
512
+ onAbort = () => reject(signal.reason ?? new Error('aborted'))
513
+ signal.addEventListener('abort', onAbort, { once: true })
514
+ })
515
+ try {
516
+ await Promise.race([Promise.resolve().then(execute), aborted])
517
+ } finally {
518
+ if (onAbort) signal.removeEventListener('abort', onAbort)
519
+ }
520
+ }
521
+
522
+ function promptUsage(prompts: readonly BenchPromptResult[]): ReturnType<typeof sumSandboxUsage> {
523
+ return combinedUsage(prompts.map((prompt) => ({ artifact: '', ok: false, usage: prompt.usage })))
524
+ }
525
+
362
526
  function parseMaybeJson(value: string): unknown {
363
527
  try {
364
528
  return JSON.parse(value) as unknown
@@ -451,15 +615,20 @@ async function loopedShot(
451
615
  return {
452
616
  artifact: completed.at(-1)?.artifact ?? '',
453
617
  ok: false,
618
+ execution: pendingShot ? { phase: 'unknown', terminalOutcome: 'unknown' } : completed.at(-1)?.execution,
619
+ artifactAvailable: false,
454
620
  usage: combinedUsage(pendingShot ? [...completed, { artifact: '', ok: false }] : completed),
455
621
  events: completed.flatMap((shot) => shot.events ?? []),
622
+ prompts: completed.flatMap((shot) => shot.prompts ?? []),
456
623
  detail: err instanceof Error ? err.message : String(err),
457
624
  }
458
625
  }
459
626
 
460
627
  const best = result.rounds.reduce((winner, candidate) => {
461
- if (shots.get(candidate.round)?.ok !== true) return winner
462
- if (shots.get(winner.round)?.ok !== true) return candidate
628
+ const candidateShot = shots.get(candidate.round)
629
+ const winnerShot = shots.get(winner.round)
630
+ const rank = (shot: BenchShotResult | undefined) => shot?.ok ? 2 : shot?.artifactAvailable ? 1 : 0
631
+ if (rank(candidateShot) !== rank(winnerShot)) return rank(candidateShot) > rank(winnerShot) ? candidate : winner
463
632
  const a = scores.get(winner.round)
464
633
  const b = scores.get(candidate.round)
465
634
  if (!a) return candidate
@@ -472,8 +641,11 @@ async function loopedShot(
472
641
  return {
473
642
  artifact: best.artifact,
474
643
  ok: shots.get(best.round)?.ok === true && best.artifact.trim().length > 0,
644
+ execution: shots.get(best.round)?.execution,
645
+ artifactAvailable: shots.get(best.round)?.artifactAvailable,
475
646
  usage: combinedUsage([...shots.values()]),
476
647
  events: [...shots.values()].flatMap((shot) => shot.events ?? []),
648
+ prompts: [...shots.values()].flatMap((shot) => shot.prompts ?? []),
477
649
  detail: JSON.stringify({
478
650
  mode: 'refine-loop',
479
651
  attempts: result.rounds.length,
@@ -588,6 +760,7 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
588
760
  const startedAt = Date.now()
589
761
  let result: BenchCellTaskResult
590
762
  let out: BenchShotResult | undefined
763
+ let invoked = false
591
764
  try {
592
765
  opts.signal?.throwIfAborted()
593
766
  const shotInput = {
@@ -603,7 +776,9 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
603
776
  ...(opts.timeoutMs ? { timeoutMs: opts.timeoutMs } : {}),
604
777
  ...(opts.signal ? { signal: opts.signal } : {}),
605
778
  ...(opts.resolveClient ? { resolveClient: opts.resolveClient } : {}),
779
+ ...(opts.execute ? { execute: opts.execute } : {}),
606
780
  }
781
+ invoked = true
607
782
  out = loopAttempts > 1 ? await loopedShot(shotInput, shot, loopAttempts) : await shot(shotInput)
608
783
  const score: BenchScore = await job.adapter.judge(job.task, out.artifact)
609
784
  result = {
@@ -614,15 +789,17 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
614
789
  resolved: out.ok && score.resolved,
615
790
  score: out.ok ? score.score : 0,
616
791
  ok: out.ok,
792
+ execution: out.execution ?? { phase: out.ok ? 'started' : 'unknown', terminalOutcome: out.ok ? 'succeeded' : 'unknown' },
793
+ measurement: (out.artifactAvailable ?? out.ok) ? 'available' : 'unavailable',
617
794
  ...(out.detail ?? score.detail ? { detail: combineDetails(out.detail, score.detail) } : {}),
618
795
  wallMs: Date.now() - startedAt,
619
796
  artifact: out.artifact,
620
797
  ...(out.usage === undefined ? {} : { usage: out.usage }),
621
798
  ...(out.events === undefined ? {} : { events: out.events }),
799
+ ...(out.prompts === undefined ? {} : { prompts: out.prompts }),
622
800
  }
623
801
  } catch (err) {
624
- // A thrown shot/judge is infra error for THIS cell-task: ok=false excludes it from the
625
- // resolve denominator (never a silent 0% that hides a harness outage).
802
+ // Missing results do not prove that dispatch or paid inference never occurred.
626
803
  result = {
627
804
  benchmark: job.benchmark,
628
805
  cell: job.cell.label,
@@ -631,11 +808,17 @@ export async function runBenchmarks(opts: RunBenchmarksOptions): Promise<RunBenc
631
808
  resolved: false,
632
809
  score: 0,
633
810
  ok: false,
811
+ execution: out?.execution ?? {
812
+ phase: !invoked ? 'not-started' : out?.ok ? 'started' : 'unknown',
813
+ terminalOutcome: out?.ok ? 'succeeded' : 'unknown',
814
+ },
815
+ measurement: 'unavailable',
634
816
  detail: err instanceof Error ? err.message.slice(0, 200) : String(err),
635
817
  wallMs: Date.now() - startedAt,
636
818
  ...(out === undefined ? {} : { artifact: out.artifact }),
637
819
  ...(out?.usage === undefined ? {} : { usage: out.usage }),
638
820
  ...(out?.events === undefined ? {} : { events: out.events }),
821
+ ...(out?.prompts === undefined ? {} : { prompts: out.prompts }),
639
822
  }
640
823
  }
641
824
  void index
@@ -660,7 +843,7 @@ function aggregate(perTask: readonly BenchCellTaskResult[]): BenchLeaderboardRow
660
843
  const key = `${r.benchmark}\u0000${r.cell}`
661
844
  const e = byKey.get(key) ?? { benchmark: r.benchmark, cell: r.cell, n: 0, resolved: 0, errored: 0, scoreSum: 0 }
662
845
  e.n += 1
663
- if (!r.ok) e.errored += 1
846
+ if ((r.measurement ?? (r.ok ? 'available' : 'unavailable')) === 'unavailable') e.errored += 1
664
847
  else {
665
848
  if (r.resolved) e.resolved += 1
666
849
  e.scoreSum += r.score