experimental-a2 0.12.0 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/ai-server.ts CHANGED
@@ -9,7 +9,7 @@
9
9
  // ordered; parallel consumption would corrupt progress sequence and state.
10
10
 
11
11
  import { AsyncLocalStorage } from 'node:async_hooks'
12
- import { convertToModelMessages } from 'ai'
12
+ import { asSchema, convertToModelMessages } from 'ai'
13
13
  import type {
14
14
  FinishReason,
15
15
  Instructions,
@@ -22,6 +22,7 @@ import type {
22
22
  UIMessageChunk,
23
23
  } from 'ai'
24
24
  import { generateAISDKStep, type AISDKStepSettings } from './ai-sdk-step.ts'
25
+ import { gatewayModelId, readModelLimits } from './ai-model-metadata.ts'
25
26
  import type {
26
27
  AIEventDefs,
27
28
  AIState,
@@ -116,6 +117,7 @@ export type AgentGenerateContext<
116
117
  generationId: string
117
118
  responseMessageId: string
118
119
  messages: M[]
120
+ modelMessages: ModelMessage[]
119
121
  state: AIState<M>
120
122
  history: ContractEvent<D>[]
121
123
  signal: AbortSignal
@@ -161,13 +163,29 @@ export type CompactionPolicy<
161
163
  D extends AIEventDefs<M> & EventDefs,
162
164
  > = {
163
165
  shouldCompact(
164
- context: AgentResolverContext<M, D> & { messages: M[] },
166
+ context: AgentResolverContext<M, D> & {
167
+ messages: M[]
168
+ modelMessages: ModelMessage[]
169
+ },
165
170
  ): boolean | Promise<boolean>
166
171
  compact(
167
- context: AgentResolverContext<M, D> & { messages: M[] },
172
+ context: AgentResolverContext<M, D> & {
173
+ messages: M[]
174
+ modelMessages: ModelMessage[]
175
+ },
168
176
  ): M[] | Promise<M[]>
169
177
  }
170
178
 
179
+ export type AutomaticCompaction = {
180
+ thresholdTokens?: number
181
+ instructions?: string
182
+ }
183
+
184
+ export type CompactionOptions<
185
+ M extends UIMessage,
186
+ D extends AIEventDefs<M> & EventDefs,
187
+ > = false | AutomaticCompaction | CompactionPolicy<M, D>
188
+
171
189
  export type AgentGenerationSettings<T extends ToolSet> = AISDKStepSettings<T>
172
190
 
173
191
  export type CreateHandlersOptions<
@@ -196,7 +214,9 @@ export type CreateHandlersOptions<
196
214
  * The custom stream owns its metadata chunks.
197
215
  */
198
216
  generate?: AgentGenerate<M, D, T>
199
- compaction?: CompactionPolicy<M, D>
217
+ compaction?:
218
+ | CompactionOptions<M, D>
219
+ | ((context: AgentResolverContext<M, D>) => CompactionOptions<M, D>)
200
220
  /** Progress is durably flushed at either limit, whichever is reached first. */
201
221
  progress?: { maxChunks?: number; maxDelayMs?: number }
202
222
  }
@@ -210,13 +230,35 @@ export type CreateAgentServerOptions<
210
230
  handlers?: ServerOptions<D>['handlers']
211
231
  }
212
232
 
213
- const resolve = async <T, C>(
233
+ const checkAbort = (signal: AbortSignal): boolean => {
234
+ if (!signal.aborted) return false
235
+ const reason: unknown = signal.reason
236
+ if (
237
+ reason instanceof A2Error &&
238
+ (reason.code === 'CLAIM_EXPIRED' || reason.code === 'SUPERSEDED_ATTEMPT')
239
+ ) {
240
+ throw reason
241
+ }
242
+ return true
243
+ }
244
+
245
+ const resolve = async <T, C extends { signal: AbortSignal }>(
214
246
  value: Resolvable<T, C>,
215
247
  context: C,
216
- ): Promise<T> =>
217
- typeof value === 'function'
218
- ? await (value as (context: C) => T | Promise<T>)(context)
219
- : value
248
+ ): Promise<T> => {
249
+ let result: T
250
+ try {
251
+ result =
252
+ typeof value === 'function'
253
+ ? await (value as (context: C) => T | Promise<T>)(context)
254
+ : value
255
+ } catch (error) {
256
+ checkAbort(context.signal)
257
+ throw error
258
+ }
259
+ checkAbort(context.signal)
260
+ return result
261
+ }
220
262
 
221
263
  const modelName = (model: LanguageModel): string => {
222
264
  if (typeof model === 'string') return model
@@ -265,10 +307,12 @@ const isBoundaryChunk = (chunk: UIMessageChunk | undefined): boolean =>
265
307
  chunk.type === 'abort' ||
266
308
  chunk.type === 'error')
267
309
 
310
+ type StreamRead<T> = Awaited<ReturnType<ReadableStreamDefaultReader<T>['read']>>
311
+
268
312
  const nextOrTimer = async <T>(
269
- next: Promise<IteratorResult<T>>,
313
+ next: Promise<StreamRead<T>>,
270
314
  delayMs: number,
271
- ): Promise<{ type: 'next'; result: IteratorResult<T> } | { type: 'timer' }> => {
315
+ ): Promise<{ type: 'next'; result: StreamRead<T> } | { type: 'timer' }> => {
272
316
  let timer: ReturnType<typeof setTimeout> | undefined
273
317
  try {
274
318
  return await Promise.race([
@@ -305,6 +349,7 @@ async function generateWithAISDK<
305
349
  model: context.model,
306
350
  tools: context.tools,
307
351
  messages: context.messages,
352
+ modelMessages: context.modelMessages,
308
353
  responseMessageId: context.responseMessageId,
309
354
  abortSignal: context.signal,
310
355
  ...(context.instructions === undefined
@@ -333,11 +378,11 @@ async function* consumeGeneration(options: {
333
378
  const maxChunks = options.progress?.maxChunks ?? 16
334
379
  const maxDelayMs = options.progress?.maxDelayMs ?? 30
335
380
  let lastFlush = Date.now()
336
- const chunks = options.source.stream[Symbol.asyncIterator]()
337
- let pendingChunk = chunks.next()
381
+ const reader = options.source.stream.getReader()
382
+ let pendingChunk = reader.read()
338
383
  try {
339
384
  for (;;) {
340
- let result: IteratorResult<UIMessageChunk>
385
+ let result: StreamRead<UIMessageChunk>
341
386
  if (pending.length > 0) {
342
387
  const remaining = Math.max(0, maxDelayMs - (Date.now() - lastFlush))
343
388
  const outcome = await nextOrTimer(pendingChunk, remaining)
@@ -355,7 +400,7 @@ async function* consumeGeneration(options: {
355
400
  pending.push(chunk)
356
401
  if (chunk.type === 'finish') streamedFinishReason = chunk.finishReason
357
402
  if (chunk.type === 'error') streamedError = chunk.errorText
358
- pendingChunk = chunks.next()
403
+ pendingChunk = reader.read()
359
404
 
360
405
  if (pending.length >= maxChunks || isBoundaryChunk(pending.at(-1))) {
361
406
  yield { type: 'progress', chunks: pending.splice(0) }
@@ -366,6 +411,15 @@ async function* consumeGeneration(options: {
366
411
  if (pending.length > 0)
367
412
  yield { type: 'progress', chunks: pending.splice(0) }
368
413
  throw error
414
+ } finally {
415
+ void pendingChunk.catch(() => {})
416
+ try {
417
+ await reader.cancel()
418
+ } catch {
419
+ // Cleanup must not replace the error that ended generation.
420
+ } finally {
421
+ reader.releaseLock()
422
+ }
369
423
  }
370
424
 
371
425
  let completion: GenerationCompletion | undefined
@@ -398,32 +452,249 @@ const replay = <M extends UIMessage, D extends AIEventDefs<M> & EventDefs>(
398
452
  return state
399
453
  }
400
454
 
401
- const contextMessages = <M extends UIMessage>(state: AIState<M>): M[] => {
455
+ const summarize = async <
456
+ M extends UIMessage,
457
+ D extends AIEventDefs<M> & EventDefs,
458
+ T extends ToolSet,
459
+ >(options: {
460
+ context: AgentGenerateContext<M, D, T>
461
+ instructions: string | undefined
462
+ }): Promise<{ summary: string; usage?: LanguageModelUsage }> => {
463
+ const instruction = [
464
+ 'Summarize this conversation so you can continue the task after the earlier context is replaced by your summary.',
465
+ 'Preserve the objective, constraints, decisions, exact identifiers, completed actions and their results, unresolved questions, and next steps. Distinguish observations from assumptions.',
466
+ 'Respond only with the summary as text. Do not call any tools or continue working on the task.',
467
+ options.instructions,
468
+ ]
469
+ .filter(Boolean)
470
+ .join('\n')
471
+ const generation = { ...options.context.generation }
472
+ for (const key of Object.keys(generation)) {
473
+ if (
474
+ key.startsWith('on') ||
475
+ key.startsWith('experimental_on') ||
476
+ [
477
+ 'repairToolCall',
478
+ 'experimental_repairToolCall',
479
+ 'experimental_refineToolInput',
480
+ 'experimental_transform',
481
+ 'toolApproval',
482
+ ].includes(key)
483
+ ) {
484
+ Reflect.deleteProperty(generation, key)
485
+ }
486
+ }
487
+ const tools = Object.fromEntries(
488
+ Object.entries(options.context.tools).map(([name, tool]) => {
489
+ const definition = { ...tool, needsApproval: false }
490
+ for (const key of [
491
+ 'execute',
492
+ 'onInputStart',
493
+ 'onInputDelta',
494
+ 'onInputAvailable',
495
+ ]) {
496
+ Reflect.deleteProperty(definition, key)
497
+ }
498
+ return [name, definition]
499
+ }),
500
+ ) as T
501
+ const source = await generateWithAISDK(
502
+ {
503
+ ...options.context,
504
+ generation,
505
+ tools,
506
+ modelMessages: [
507
+ ...options.context.modelMessages,
508
+ { role: 'user', content: instruction },
509
+ ],
510
+ responseMessageId: `${options.context.generationId}:summary`,
511
+ },
512
+ undefined,
513
+ )
514
+ let summary = ''
515
+ let calledTool = false
516
+ let finish: AgentGenerationFinish | undefined
517
+ for await (const update of consumeGeneration({ source })) {
518
+ options.context.signal.throwIfAborted()
519
+ if (update.type === 'finish') {
520
+ finish = update
521
+ continue
522
+ }
523
+ for (const chunk of update.chunks) {
524
+ if (chunk.type === 'text-delta') summary += chunk.delta
525
+ if (chunk.type.startsWith('tool-')) calledTool = true
526
+ }
527
+ }
528
+ if (
529
+ calledTool ||
530
+ summary.trim().length === 0 ||
531
+ finish?.finishReason !== 'stop'
532
+ ) {
533
+ throw new Error(
534
+ 'compaction must produce a complete text summary without calling tools',
535
+ )
536
+ }
537
+ return {
538
+ summary,
539
+ ...(finish.usage === undefined ? {} : { usage: finish.usage }),
540
+ }
541
+ }
542
+
543
+ const activeCompaction = <M extends UIMessage>(
544
+ state: AIState<M>,
545
+ ): AIState<M>['compaction'] => {
402
546
  const compaction = state.compaction
403
- if (compaction?.status !== 'completed' || !compaction.messages) {
404
- return state.messages
547
+ return compaction?.status === 'completed' &&
548
+ compaction.messages !== undefined &&
549
+ state.messages.some((message) => message.id === compaction.throughMessageId)
550
+ ? compaction
551
+ : null
552
+ }
553
+
554
+ const contextMessages = <
555
+ M extends UIMessage,
556
+ D extends AIEventDefs<M> & EventDefs,
557
+ >(options: {
558
+ agent: AgentDefinition<M, D>
559
+ state: AIState<M>
560
+ history: ContractEvent<D>[]
561
+ }): M[] => {
562
+ const { agent, state, history } = options
563
+ const compaction = activeCompaction(state)
564
+ if (compaction === null) return state.messages
565
+ const retained = new Set(compaction.retainedMessageIds ?? [])
566
+ const startedIndex = history.find(
567
+ (event) =>
568
+ event.type === 'ai.generation.started' &&
569
+ (event.payload as GenerationStartedPayload).generationId ===
570
+ compaction.generationId,
571
+ )?.index
572
+ const throughIndex =
573
+ compaction.throughIndex ??
574
+ (startedIndex === undefined ? undefined : startedIndex - 1)
575
+ if (throughIndex !== undefined) {
576
+ const prefix = history.filter((event) => event.index <= throughIndex)
577
+ for (const queued of queuedMessagesAt(prefix))
578
+ retained.add(queued.messageId)
579
+ let context = replay(agent, prefix)
580
+ context = {
581
+ ...context,
582
+ messages: [
583
+ ...compaction.messages!,
584
+ ...context.messages.filter((message) => retained.has(message.id)),
585
+ ],
586
+ activeProjection: null,
587
+ }
588
+ for (const event of history) {
589
+ if (event.index > throughIndex)
590
+ context = agent.reducer.fold(context, event)
591
+ }
592
+ return context.messages
405
593
  }
406
594
  const boundary = state.messages.findIndex(
407
595
  (message) => message.id === compaction.throughMessageId,
408
596
  )
409
- const retained = new Set(compaction.retainedMessageIds ?? [])
410
- return boundary === -1
411
- ? state.messages
412
- : [
413
- ...compaction.messages,
414
- ...state.messages
415
- .slice(0, boundary + 1)
416
- .filter((message) => retained.has(message.id)),
417
- ...state.messages.slice(boundary + 1),
418
- ]
597
+ return [
598
+ ...compaction.messages!,
599
+ ...state.messages
600
+ .slice(0, boundary + 1)
601
+ .filter((message) => retained.has(message.id)),
602
+ ...state.messages.slice(boundary + 1),
603
+ ]
419
604
  }
420
605
 
421
- const activeContextMessages = <M extends UIMessage>(
422
- state: AIState<M>,
423
- coordinator: AICoordinatorState,
424
- ): M[] => {
425
- const queued = new Set(coordinator.queued.map((item) => item.messageId))
426
- return contextMessages(state).filter((message) => !queued.has(message.id))
606
+ const activeContextMessages = <
607
+ M extends UIMessage,
608
+ D extends AIEventDefs<M> & EventDefs,
609
+ >(options: {
610
+ agent: AgentDefinition<M, D>
611
+ state: AIState<M>
612
+ history: ContractEvent<D>[]
613
+ coordinator: AICoordinatorState
614
+ }): M[] => {
615
+ const queued = new Set(
616
+ options.coordinator.queued.map((item) => item.messageId),
617
+ )
618
+ return contextMessages(options).filter((message) => !queued.has(message.id))
619
+ }
620
+
621
+ const modelContext = async <M extends UIMessage>(options: {
622
+ messages: M[]
623
+ state: AIState<M>
624
+ tools: ToolSet
625
+ }): Promise<ModelMessage[]> => {
626
+ const messages = await convertToModelMessages(options.messages, {
627
+ tools: options.tools,
628
+ })
629
+ const summary = activeCompaction(options.state)?.summary
630
+ return summary === undefined
631
+ ? messages
632
+ : [{ role: 'user', content: summary }, ...messages]
633
+ }
634
+
635
+ const estimateInputTokens = async (options: {
636
+ messages: ModelMessage[]
637
+ instructions: Instructions | undefined
638
+ tools: ToolSet
639
+ }): Promise<number> => {
640
+ const tools = await Promise.all(
641
+ Object.entries(options.tools).map(async ([name, tool]) => ({
642
+ name,
643
+ description: tool.description,
644
+ inputSchema: await asSchema(tool.inputSchema).jsonSchema,
645
+ })),
646
+ )
647
+ return Math.ceil(
648
+ new TextEncoder().encode(
649
+ JSON.stringify({
650
+ messages: options.messages,
651
+ instructions: options.instructions,
652
+ tools,
653
+ }),
654
+ ).byteLength / 4,
655
+ )
656
+ }
657
+
658
+ const measuredInputTokens = <D extends EventDefs>(options: {
659
+ estimate: number
660
+ model: string
661
+ history: ContractEvent<D>[]
662
+ }): number => {
663
+ for (const event of options.history.toReversed()) {
664
+ if (
665
+ event.type === 'ai.compaction.completed' ||
666
+ event.type === 'ai.retry.requested'
667
+ )
668
+ break
669
+ if (event.type !== 'ai.generation.completed') continue
670
+ const completed = event.payload as GenerationCompletedPayload
671
+ const started = options.history.find(
672
+ (candidate) =>
673
+ candidate.type === 'ai.generation.started' &&
674
+ (candidate.payload as GenerationStartedPayload).generationId ===
675
+ completed.generationId,
676
+ )
677
+ const input = completed.usage?.inputTokens
678
+ const estimate = completed.inputTokenEstimate
679
+ if (
680
+ (started?.payload as GenerationStartedPayload | undefined)?.model !==
681
+ options.model
682
+ )
683
+ break
684
+ if (
685
+ input === undefined ||
686
+ estimate === undefined ||
687
+ !Number.isFinite(input) ||
688
+ input < 0 ||
689
+ options.estimate < estimate
690
+ )
691
+ break
692
+ return Math.max(
693
+ options.estimate,
694
+ Math.ceil(input + options.estimate - estimate),
695
+ )
696
+ }
697
+ return options.estimate
427
698
  }
428
699
 
429
700
  const generationRequestId = (generationId: string): string | undefined => {
@@ -772,6 +1043,62 @@ function withoutGenerationLifecycle<D extends EventDefs>(
772
1043
  })
773
1044
  }
774
1045
 
1046
+ const validateCompaction = <
1047
+ M extends UIMessage,
1048
+ D extends AIEventDefs<M> & EventDefs,
1049
+ T extends ToolSet,
1050
+ >(options: {
1051
+ compaction: CompactionOptions<M, D>
1052
+ explicit: boolean
1053
+ generation: AgentGenerationSettings<T> | undefined
1054
+ tools: T | undefined
1055
+ }): CompactionOptions<M, D> => {
1056
+ let compaction = options.compaction
1057
+ if (
1058
+ compaction !== false &&
1059
+ (typeof compaction !== 'object' || compaction === null)
1060
+ ) {
1061
+ throw new TypeError('compaction must resolve to false or a policy object')
1062
+ }
1063
+ if (compaction !== false && 'then' in compaction) {
1064
+ void Promise.resolve(compaction).catch(() => {})
1065
+ throw new TypeError('compaction options must resolve synchronously')
1066
+ }
1067
+ if (compaction !== false && !('shouldCompact' in compaction)) {
1068
+ if (
1069
+ compaction.thresholdTokens !== undefined &&
1070
+ (!Number.isSafeInteger(compaction.thresholdTokens) ||
1071
+ compaction.thresholdTokens < 1)
1072
+ ) {
1073
+ throw new TypeError(
1074
+ 'compaction.thresholdTokens must be a positive safe integer',
1075
+ )
1076
+ }
1077
+ const choice = options.generation?.toolChoice
1078
+ const unsupported =
1079
+ options.generation?.output !== undefined
1080
+ ? 'structured output'
1081
+ : choice !== undefined && choice !== 'auto' && choice !== 'none'
1082
+ ? 'forced tool choice'
1083
+ : Object.values(options.tools ?? {}).some(
1084
+ (tool) =>
1085
+ tool.type === 'provider' && tool.isProviderExecuted === true,
1086
+ )
1087
+ ? 'provider-executed tools'
1088
+ : undefined
1089
+ if (unsupported !== undefined) {
1090
+ if (options.explicit) {
1091
+ throw new TypeError(
1092
+ `automatic compaction does not support ${unsupported}; use a custom policy or compaction: false`,
1093
+ )
1094
+ }
1095
+ compaction = false
1096
+ }
1097
+ }
1098
+
1099
+ return compaction
1100
+ }
1101
+
775
1102
  /**
776
1103
  * Build the ordinary A2 handler table for the built-in agent protocol.
777
1104
  * Application handlers can be spread beside this table.
@@ -808,8 +1135,20 @@ export function createHandlers<
808
1135
  throw new TypeError('maxSteps must be a positive integer or Infinity')
809
1136
  }
810
1137
 
1138
+ const configuredCompaction =
1139
+ options.compaction ?? (options.generate === undefined ? {} : false)
1140
+ const staticCompaction =
1141
+ typeof configuredCompaction === 'function'
1142
+ ? undefined
1143
+ : validateCompaction({
1144
+ compaction: configuredCompaction,
1145
+ explicit: options.compaction !== undefined,
1146
+ generation: options.generation,
1147
+ tools: options.tools,
1148
+ })
1149
+
811
1150
  const tools = options.tools ?? ({} as T)
812
- const generation = options.generation ?? {}
1151
+ const generation: AgentGenerationSettings<T> = options.generation ?? {}
813
1152
  const maxSteps = options.maxSteps ?? Number.POSITIVE_INFINITY
814
1153
  const coordinator = aiCoordinatorReducer(options.agent.contract)
815
1154
  const promptCache = new Map<string, Promise<ModelMessage[]>>()
@@ -956,39 +1295,37 @@ export function createHandlers<
956
1295
  if (cached) return cached
957
1296
  const computation = (async () => {
958
1297
  const history = await readHistory()
959
- const coordinatorState = coordinatorStateAt(history)
960
- const compaction = history.find(
961
- (event) =>
962
- event.type === 'ai.compaction.completed' &&
963
- (event.payload as CompactionCompletedPayload<M>).generationId ===
964
- call.generationId,
965
- )
966
- if (compaction !== undefined) {
967
- const messages = (
968
- compaction.payload as CompactionCompletedPayload<M>
969
- ).messages.filter(
970
- (message) =>
971
- !coordinatorState.queued.some(
972
- (queued) => queued.messageId === message.id,
973
- ),
974
- )
975
- return convertToModelMessages(messages, { tools })
976
- }
977
1298
  const started = history.find(
978
1299
  (event) =>
979
1300
  event.type === 'ai.generation.started' &&
980
1301
  (event.payload as GenerationStartedPayload).generationId ===
981
1302
  call.generationId,
982
1303
  )
983
- const frontier = started?.index ?? Number.POSITIVE_INFINITY
984
- const state = replay(
985
- options.agent,
986
- history.filter((event) => event.index < frontier),
987
- )
988
- return convertToModelMessages(
989
- activeContextMessages(state, coordinatorState),
990
- { tools },
1304
+ const startedPayload = started?.payload as
1305
+ GenerationStartedPayload | undefined
1306
+ const frontier =
1307
+ startedPayload?.promptThroughIndex ??
1308
+ (started?.index ?? Number.POSITIVE_INFINITY) - 1
1309
+ const promptHistory = history.filter(
1310
+ (event) =>
1311
+ event.index <= frontier ||
1312
+ ((event.type === 'ai.generation.started' ||
1313
+ event.type === 'ai.compaction.requested' ||
1314
+ event.type === 'ai.compaction.completed') &&
1315
+ (event.payload as { generationId: string }).generationId ===
1316
+ call.generationId),
991
1317
  )
1318
+ const state = replay(options.agent, promptHistory)
1319
+ return modelContext({
1320
+ messages: activeContextMessages({
1321
+ agent: options.agent,
1322
+ state,
1323
+ history: promptHistory,
1324
+ coordinator: coordinatorStateAt(promptHistory),
1325
+ }),
1326
+ state,
1327
+ tools,
1328
+ })
992
1329
  })()
993
1330
  promptCache.set(key, computation)
994
1331
  void computation.catch(() => {
@@ -1003,7 +1340,7 @@ export function createHandlers<
1003
1340
  error: unknown,
1004
1341
  ): AppendInput<D> | void => {
1005
1342
  const schedulerFailure = consumeSchedulerSendFailure(error)
1006
- if (ctx.signal.aborted) return
1343
+ if (checkAbort(ctx.signal)) return
1007
1344
  if (schedulerFailure === 'retryable') throw error
1008
1345
  return resultEvent(call, 'execution:error', {
1009
1346
  error: errorMessage(error),
@@ -1037,40 +1374,42 @@ export function createHandlers<
1037
1374
  }
1038
1375
  let last: unknown
1039
1376
  let sequence = 0
1040
- for (;;) {
1041
- let result: IteratorResult<unknown>
1042
- try {
1043
- result = await iterator.next()
1044
- } catch (error) {
1045
- return toolExecutionFailure(ctx, call, error)
1046
- }
1047
- if (result.done) break
1048
- if (ctx.signal.aborted) {
1377
+ let done = false
1378
+ try {
1379
+ for (;;) {
1380
+ let result: IteratorResult<unknown>
1049
1381
  try {
1050
- await iterator.return?.()
1051
- } catch {
1052
- // The aborted handler cannot record an iterator cleanup failure.
1382
+ result = await iterator.next()
1383
+ } catch (error) {
1384
+ return toolExecutionFailure(ctx, call, error)
1053
1385
  }
1054
- return
1055
- }
1056
- last = result.value
1057
- try {
1386
+ if (checkAbort(ctx.signal)) return
1387
+ if (result.done) {
1388
+ done = true
1389
+ break
1390
+ }
1391
+ last = result.value
1058
1392
  await ctx.session.append(
1059
1393
  `tool:${call.toolCallId}:preliminary:${sequence}`,
1060
- resultEvent(call, `execution:${sequence}:preliminary`, {
1061
- output: result.value,
1062
- preliminary: true,
1063
- }),
1394
+ resultEvent(
1395
+ call,
1396
+ `execution:${ctx.attempt}:${sequence}:preliminary`,
1397
+ {
1398
+ output: result.value,
1399
+ preliminary: true,
1400
+ },
1401
+ ),
1064
1402
  )
1065
- } catch (error) {
1403
+ sequence += 1
1404
+ }
1405
+ } finally {
1406
+ if (!done) {
1066
1407
  try {
1067
1408
  await iterator.return?.()
1068
1409
  } catch {
1069
- // Preserve the append failure that makes the durable handler retry.
1410
+ // Preserve the failure or cancellation that ended execution.
1070
1411
  }
1071
- throw error
1072
1412
  }
1073
- sequence += 1
1074
1413
  }
1075
1414
  return resultEvent(
1076
1415
  call,
@@ -1095,7 +1434,7 @@ export function createHandlers<
1095
1434
  call,
1096
1435
  ctx.session.history,
1097
1436
  )
1098
- if (ctx.signal.aborted) return
1437
+ if (checkAbort(ctx.signal)) return
1099
1438
  const scope: AmbientToolScope = {
1100
1439
  contract: options.agent.contract,
1101
1440
  context: ctx,
@@ -1262,7 +1601,7 @@ export function createHandlers<
1262
1601
  history,
1263
1602
  replacedGenerationIds,
1264
1603
  )
1265
- let state = replay(options.agent, promptHistory)
1604
+ const state = replay(options.agent, promptHistory)
1266
1605
  const resolverContext: AgentResolverContext<M, D> = {
1267
1606
  event: ctx.event,
1268
1607
  state,
@@ -1270,138 +1609,289 @@ export function createHandlers<
1270
1609
  signal: ctx.signal,
1271
1610
  }
1272
1611
 
1273
- const resolvedModel = await resolve(options.model, resolverContext)
1274
- if (resolvedModel === undefined) {
1275
- throw new TypeError('the model resolver returned undefined')
1276
- }
1277
- if (ctx.signal.aborted) return
1278
-
1279
- if (request.reason === 'tool' && responseStepCount >= maxSteps) {
1280
- const source = sourceCoordinatorState.response?.generation
1281
- if (source === undefined) return
1282
- return {
1283
- type: 'ai.generation.failed',
1284
- id: `${requestId}:step-limit`,
1285
- payload: {
1286
- requestId: source.requestId,
1287
- messageId: source.messageId,
1288
- generationId: source.generationId,
1289
- responseMessageId,
1290
- error: `agent exceeded the ${maxSteps}-step limit`,
1291
- stepLimit: true,
1292
- },
1293
- } as AppendInput<D>
1294
- }
1612
+ let generationStarted = false
1613
+ try {
1614
+ const resolvedModel = await resolve(options.model, resolverContext)
1615
+ if (resolvedModel === undefined) {
1616
+ throw new TypeError('the model resolver returned undefined')
1617
+ }
1618
+ if (checkAbort(ctx.signal)) return
1619
+ const resolvedInstructions =
1620
+ options.instructions === undefined
1621
+ ? undefined
1622
+ : await resolve(options.instructions, resolverContext)
1623
+ if (checkAbort(ctx.signal)) return
1624
+ const compaction =
1625
+ typeof configuredCompaction === 'function'
1626
+ ? validateCompaction({
1627
+ compaction: configuredCompaction(resolverContext),
1628
+ explicit: true,
1629
+ generation: options.generation,
1630
+ tools: options.tools,
1631
+ })
1632
+ : staticCompaction!
1633
+ if (checkAbort(ctx.signal)) return
1634
+
1635
+ if (request.reason === 'tool' && responseStepCount >= maxSteps) {
1636
+ const source = sourceCoordinatorState.response?.generation
1637
+ if (source === undefined) return
1638
+ return {
1639
+ type: 'ai.generation.failed',
1640
+ id: `${requestId}:step-limit`,
1641
+ payload: {
1642
+ requestId: source.requestId,
1643
+ messageId: source.messageId,
1644
+ generationId: source.generationId,
1645
+ responseMessageId,
1646
+ error: `agent exceeded the ${maxSteps}-step limit`,
1647
+ stepLimit: true,
1648
+ },
1649
+ } as AppendInput<D>
1650
+ }
1295
1651
 
1296
- const started: GenerationStartedPayload = {
1297
- requestId,
1298
- messageId: request.messageId,
1299
- generationId,
1300
- responseMessageId,
1301
- attempt,
1302
- model: modelName(resolvedModel),
1303
- }
1304
- const startEvents: AppendInput<D>[] = []
1305
- if (previous) {
1306
- const payload = previous.payload as GenerationStartedPayload
1307
- const superseded: GenerationFailedPayload = {
1652
+ const started: GenerationStartedPayload = {
1308
1653
  requestId,
1309
- messageId: payload.messageId,
1310
- generationId: payload.generationId,
1311
- responseMessageId: payload.responseMessageId,
1312
- error: 'generation attempt was superseded after an incomplete run',
1313
- superseded: true,
1654
+ messageId: request.messageId,
1655
+ generationId,
1656
+ responseMessageId,
1657
+ attempt,
1658
+ model: modelName(resolvedModel),
1659
+ promptThroughIndex: history.at(-1)?.index ?? ctx.event.index,
1660
+ }
1661
+ const catalogModelId = gatewayModelId(resolvedModel)
1662
+ const gatewayOptions = options.generation?.providerOptions?.['gateway']
1663
+ const usesFallbackModels =
1664
+ Array.isArray(gatewayOptions?.['models']) &&
1665
+ gatewayOptions['models'].length > 0
1666
+ const discoversMetadata =
1667
+ compaction !== false &&
1668
+ !('shouldCompact' in compaction) &&
1669
+ compaction.thresholdTokens === undefined &&
1670
+ !usesFallbackModels &&
1671
+ catalogModelId !== undefined
1672
+ const startEvents: AppendInput<D>[] = []
1673
+ if (
1674
+ discoversMetadata &&
1675
+ !Object.hasOwn(state.modelMetadata, catalogModelId)
1676
+ ) {
1677
+ startEvents.push({
1678
+ type: 'ai.model.metadata.requested',
1679
+ id: `ai.model.metadata:${encodeURIComponent(options.agent.contract.name)}:${encodeURIComponent(ctx.event.sessionId)}:${encodeURIComponent(catalogModelId)}`,
1680
+ payload: { modelId: catalogModelId },
1681
+ } as AppendInput<D>)
1682
+ }
1683
+ if (previous) {
1684
+ const payload = previous.payload as GenerationStartedPayload
1685
+ const superseded: GenerationFailedPayload = {
1686
+ requestId,
1687
+ messageId: payload.messageId,
1688
+ generationId: payload.generationId,
1689
+ responseMessageId: payload.responseMessageId,
1690
+ error: 'generation attempt was superseded after an incomplete run',
1691
+ superseded: true,
1692
+ }
1693
+ startEvents.push({
1694
+ type: 'ai.generation.failed',
1695
+ id: `${payload.generationId}:superseded`,
1696
+ payload: superseded,
1697
+ } as AppendInput<D>)
1314
1698
  }
1315
1699
  startEvents.push({
1316
- type: 'ai.generation.failed',
1317
- id: `${payload.generationId}:superseded`,
1318
- payload: superseded,
1700
+ type: 'ai.generation.started',
1701
+ id: generationId,
1702
+ payload: started,
1319
1703
  } as AppendInput<D>)
1320
- }
1321
- startEvents.push({
1322
- type: 'ai.generation.started',
1323
- id: generationId,
1324
- payload: started,
1325
- } as AppendInput<D>)
1326
- await ctx.session.append('generation-start', ...startEvents)
1704
+ const generationHistory = [
1705
+ ...history,
1706
+ ...(await ctx.session.append('generation-start', ...startEvents)),
1707
+ ]
1708
+ generationStarted = true
1327
1709
 
1328
- const startedCoordinatorState = (
1329
- await ctx.session.state(coordinator, { through: 'latest' })
1330
- ).state
1331
- if (
1332
- startedCoordinatorState.response?.activeRequestId !== requestId ||
1333
- startedCoordinatorState.response.generation?.generationId !== generationId
1334
- ) {
1335
- return
1336
- }
1710
+ const startedCoordinatorState = (
1711
+ await ctx.session.state(coordinator, { through: 'latest' })
1712
+ ).state
1713
+ if (
1714
+ startedCoordinatorState.response?.activeRequestId !== requestId ||
1715
+ startedCoordinatorState.response.generation?.generationId !==
1716
+ generationId
1717
+ ) {
1718
+ return
1719
+ }
1337
1720
 
1338
- const promptCoordinatorState = {
1339
- ...coordinatorStateAt(history),
1340
- queued: queuedMessagesAt(history),
1341
- }
1342
- let messages = activeContextMessages(state, promptCoordinatorState)
1343
- if (options.compaction) {
1344
- const compactionContext = { ...resolverContext, messages }
1345
- if (await options.compaction.shouldCompact(compactionContext)) {
1346
- const throughMessageId = messages.at(-1)?.id ?? request.messageId
1347
- await ctx.session.append('compaction-requested', {
1348
- type: 'ai.compaction.requested',
1349
- id: `${generationId}:compaction:requested`,
1350
- payload: { generationId, throughMessageId },
1351
- })
1352
- messages = await options.compaction.compact(compactionContext)
1353
- const completed: CompactionCompletedPayload<M> = {
1354
- generationId,
1355
- throughMessageId,
1721
+ const promptCoordinatorState = {
1722
+ ...coordinatorStateAt(history),
1723
+ queued: queuedMessagesAt(history),
1724
+ }
1725
+ const messages = activeContextMessages({
1726
+ agent: options.agent,
1727
+ state,
1728
+ history: promptHistory,
1729
+ coordinator: promptCoordinatorState,
1730
+ })
1731
+ let modelMessages = await modelContext({ messages, state, tools })
1732
+ const baseContext: AgentGenerateContext<M, D, T> = {
1733
+ request: ctx.event,
1734
+ requestId,
1735
+ generationId,
1736
+ responseMessageId,
1737
+ messages,
1738
+ modelMessages,
1739
+ state,
1740
+ history,
1741
+ signal: ctx.signal,
1742
+ model: resolvedModel,
1743
+ tools,
1744
+ ...(resolvedInstructions === undefined
1745
+ ? {}
1746
+ : { instructions: resolvedInstructions }),
1747
+ generation,
1748
+ }
1749
+ const policy = compaction
1750
+ const tracksInput =
1751
+ policy !== false &&
1752
+ !('shouldCompact' in policy) &&
1753
+ (policy.thresholdTokens !== undefined || discoversMetadata)
1754
+ const metadata =
1755
+ catalogModelId === undefined
1756
+ ? undefined
1757
+ : state.modelMetadata[catalogModelId]
1758
+ const limits =
1759
+ metadata?.status === 'resolved' && !usesFallbackModels
1760
+ ? metadata.limits
1761
+ : undefined
1762
+ let inputTokenEstimate: number | undefined
1763
+ let compacted = false
1764
+ const canCompact =
1765
+ sourceCoordinatorState.response?.calls.every((call) => call.terminal) ??
1766
+ true
1767
+ if (policy && canCompact) {
1768
+ const compactionContext = {
1769
+ ...resolverContext,
1356
1770
  messages,
1357
- ...(promptCoordinatorState.queued.length === 0
1358
- ? {}
1359
- : {
1360
- retainedMessageIds: promptCoordinatorState.queued.map(
1361
- (item) => item.messageId,
1362
- ),
1363
- }),
1771
+ modelMessages,
1364
1772
  }
1365
- await ctx.session.append('compaction-completed', {
1366
- type: 'ai.compaction.completed',
1367
- id: `${generationId}:compaction:completed`,
1368
- payload: completed,
1369
- })
1370
- state = {
1371
- ...state,
1372
- compaction: { status: 'completed', ...completed },
1773
+ let shouldCompact: boolean
1774
+ if ('shouldCompact' in policy) {
1775
+ shouldCompact = await policy.shouldCompact(compactionContext)
1776
+ } else if (tracksInput) {
1777
+ inputTokenEstimate = await estimateInputTokens({
1778
+ messages: modelMessages,
1779
+ instructions: resolvedInstructions,
1780
+ tools,
1781
+ })
1782
+ const inputTokens = measuredInputTokens({
1783
+ estimate: inputTokenEstimate,
1784
+ model: modelName(resolvedModel),
1785
+ history,
1786
+ })
1787
+ const threshold =
1788
+ policy.thresholdTokens ??
1789
+ (limits === undefined
1790
+ ? undefined
1791
+ : Math.floor(
1792
+ Math.min(
1793
+ limits.contextWindow * 0.75,
1794
+ limits.contextWindow - (generation.maxOutputTokens ?? 0),
1795
+ ),
1796
+ ))
1797
+ if (
1798
+ limits !== undefined &&
1799
+ threshold !== undefined &&
1800
+ threshold <= 0
1801
+ ) {
1802
+ throw new Error(
1803
+ `output allowance exhausts the ${limits.contextWindow}-token context window`,
1804
+ )
1805
+ }
1806
+ shouldCompact = threshold !== undefined && inputTokens >= threshold
1807
+ } else {
1808
+ shouldCompact = false
1809
+ }
1810
+ if (checkAbort(ctx.signal)) return
1811
+ if (shouldCompact) {
1812
+ const throughMessageId =
1813
+ state.messages.findLast(
1814
+ (message) =>
1815
+ !promptCoordinatorState.queued.some(
1816
+ (queued) => queued.messageId === message.id,
1817
+ ),
1818
+ )?.id ?? request.messageId
1819
+ const throughIndex = history.at(-1)?.index ?? ctx.event.index
1820
+ generationHistory.push(
1821
+ ...(await ctx.session.append('compaction-requested', {
1822
+ type: 'ai.compaction.requested',
1823
+ id: `${generationId}:compaction:requested`,
1824
+ payload: { generationId, throughMessageId, throughIndex },
1825
+ })),
1826
+ )
1827
+ const result = !('shouldCompact' in policy)
1828
+ ? {
1829
+ messages: [] as M[],
1830
+ ...(await summarize({
1831
+ context: baseContext,
1832
+ instructions: policy.instructions,
1833
+ })),
1834
+ }
1835
+ : { messages: await policy.compact(compactionContext) }
1836
+ if (checkAbort(ctx.signal)) return
1837
+ const completed: CompactionCompletedPayload<M> = {
1838
+ generationId,
1839
+ throughMessageId,
1840
+ throughIndex,
1841
+ ...result,
1842
+ ...(promptCoordinatorState.queued.length === 0
1843
+ ? {}
1844
+ : {
1845
+ retainedMessageIds: promptCoordinatorState.queued.map(
1846
+ (item) => item.messageId,
1847
+ ),
1848
+ }),
1849
+ }
1850
+ generationHistory.push(
1851
+ ...(await ctx.session.append('compaction-completed', {
1852
+ type: 'ai.compaction.completed',
1853
+ id: `${generationId}:compaction:completed`,
1854
+ payload: completed,
1855
+ })),
1856
+ )
1857
+ compacted = true
1373
1858
  }
1374
1859
  }
1375
- }
1376
1860
 
1377
- const currentState = (
1378
- await ctx.session.state(options.agent.reducer, { through: 'latest' })
1379
- ).state
1380
- const currentCoordinatorState = (
1381
- await ctx.session.state(coordinator, { through: 'latest' })
1382
- ).state
1383
- const generationMessages = activeContextMessages(
1384
- currentState,
1385
- currentCoordinatorState,
1386
- )
1387
- promptCache.set(
1388
- promptCacheKey(ctx.event.sessionId, generationId),
1389
- convertToModelMessages(generationMessages, { tools }),
1390
- )
1861
+ const currentState = replay(options.agent, generationHistory)
1862
+ const currentCoordinatorState = coordinatorStateAt(generationHistory)
1863
+ const generationMessages = activeContextMessages({
1864
+ agent: options.agent,
1865
+ state: currentState,
1866
+ history: generationHistory,
1867
+ coordinator: currentCoordinatorState,
1868
+ })
1869
+ modelMessages = await modelContext({
1870
+ messages: generationMessages,
1871
+ state: currentState,
1872
+ tools,
1873
+ })
1874
+ promptCache.set(
1875
+ promptCacheKey(ctx.event.sessionId, generationId),
1876
+ Promise.resolve(modelMessages),
1877
+ )
1878
+ if (tracksInput && (inputTokenEstimate === undefined || compacted)) {
1879
+ inputTokenEstimate = await estimateInputTokens({
1880
+ messages: modelMessages,
1881
+ instructions: resolvedInstructions,
1882
+ tools,
1883
+ })
1884
+ }
1391
1885
 
1392
- let pendingToolCalls: PendingToolCall[] = []
1393
- try {
1394
- const resolvedInstructions =
1395
- options.instructions === undefined
1396
- ? undefined
1397
- : await resolve(options.instructions, resolverContext)
1398
- if (ctx.signal.aborted) return
1886
+ let pendingToolCalls: PendingToolCall[] = []
1887
+ if (checkAbort(ctx.signal)) return
1399
1888
  const generateContext: AgentGenerateContext<M, D, T> = {
1400
1889
  request: ctx.event,
1401
1890
  requestId,
1402
1891
  generationId,
1403
1892
  responseMessageId,
1404
1893
  messages: generationMessages,
1894
+ modelMessages,
1405
1895
  state: currentState,
1406
1896
  history,
1407
1897
  signal: ctx.signal,
@@ -1470,6 +1960,7 @@ export function createHandlers<
1470
1960
  messageId: request.messageId,
1471
1961
  generationId,
1472
1962
  responseMessageId,
1963
+ ...(inputTokenEstimate === undefined ? {} : { inputTokenEstimate }),
1473
1964
  ...(finish.finishReason === undefined
1474
1965
  ? {}
1475
1966
  : { finishReason: finish.finishReason }),
@@ -1513,8 +2004,11 @@ export function createHandlers<
1513
2004
  } as AppendInput<D>,
1514
2005
  ]
1515
2006
  } catch (error) {
1516
- if (error instanceof A2Error) throw error
1517
- if (ctx.signal.aborted) {
2007
+ if (error instanceof A2Error) {
2008
+ checkAbort(ctx.signal)
2009
+ throw error
2010
+ }
2011
+ if (checkAbort(ctx.signal)) {
1518
2012
  const interrupted: MessageInterruptedPayload = {
1519
2013
  messageId: responseMessageId,
1520
2014
  generationId,
@@ -1526,6 +2020,7 @@ export function createHandlers<
1526
2020
  payload: interrupted,
1527
2021
  } as AppendInput<D>
1528
2022
  }
2023
+ if (!generationStarted) throw error
1529
2024
  const failed: GenerationFailedPayload = {
1530
2025
  requestId,
1531
2026
  messageId: request.messageId,
@@ -1609,6 +2104,20 @@ export function createHandlers<
1609
2104
  }
1610
2105
 
1611
2106
  return {
2107
+ 'ai.model.metadata.requested': {
2108
+ handler: async (ctx) => {
2109
+ const limits = await readModelLimits({
2110
+ modelId: ctx.event.payload.modelId,
2111
+ signal: ctx.signal,
2112
+ })
2113
+ ctx.signal.throwIfAborted()
2114
+ return {
2115
+ type: 'ai.model.metadata.resolved',
2116
+ id: `${ctx.event.id}:resolved`,
2117
+ payload: { modelId: ctx.event.payload.modelId, limits },
2118
+ } as AppendInput<D>
2119
+ },
2120
+ },
1612
2121
  'ai.message.created': { lane: 'a2.ai.turn', handler: handleMessageCreated },
1613
2122
  'ai.retry.requested': { lane: 'a2.ai.turn', handler: handleRetry },
1614
2123
  'ai.input.responded': {