@tanstack/openai-base 0.2.1 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +121 -0
  2. package/dist/esm/adapters/chat-completions-text.d.ts +49 -21
  3. package/dist/esm/adapters/chat-completions-text.js +480 -68
  4. package/dist/esm/adapters/chat-completions-text.js.map +1 -1
  5. package/dist/esm/adapters/chat-completions-tool-converter.d.ts +8 -4
  6. package/dist/esm/adapters/chat-completions-tool-converter.js.map +1 -1
  7. package/dist/esm/adapters/responses-text.d.ts +46 -33
  8. package/dist/esm/adapters/responses-text.js +661 -142
  9. package/dist/esm/adapters/responses-text.js.map +1 -1
  10. package/dist/esm/index.d.ts +2 -9
  11. package/dist/esm/index.js +4 -16
  12. package/dist/esm/index.js.map +1 -1
  13. package/dist/esm/tools/apply-patch-tool.d.ts +2 -2
  14. package/dist/esm/tools/apply-patch-tool.js.map +1 -1
  15. package/dist/esm/tools/code-interpreter-tool.d.ts +3 -2
  16. package/dist/esm/tools/code-interpreter-tool.js.map +1 -1
  17. package/dist/esm/tools/computer-use-tool.d.ts +2 -2
  18. package/dist/esm/tools/computer-use-tool.js.map +1 -1
  19. package/dist/esm/tools/custom-tool.d.ts +2 -2
  20. package/dist/esm/tools/custom-tool.js.map +1 -1
  21. package/dist/esm/tools/file-search-tool.d.ts +2 -2
  22. package/dist/esm/tools/file-search-tool.js.map +1 -1
  23. package/dist/esm/tools/function-tool.d.ts +2 -2
  24. package/dist/esm/tools/function-tool.js.map +1 -1
  25. package/dist/esm/tools/image-generation-tool.d.ts +3 -2
  26. package/dist/esm/tools/image-generation-tool.js.map +1 -1
  27. package/dist/esm/tools/local-shell-tool.d.ts +3 -2
  28. package/dist/esm/tools/local-shell-tool.js.map +1 -1
  29. package/dist/esm/tools/mcp-tool.d.ts +3 -2
  30. package/dist/esm/tools/mcp-tool.js.map +1 -1
  31. package/dist/esm/tools/shell-tool.d.ts +2 -2
  32. package/dist/esm/tools/shell-tool.js.map +1 -1
  33. package/dist/esm/tools/web-search-preview-tool.d.ts +2 -2
  34. package/dist/esm/tools/web-search-preview-tool.js.map +1 -1
  35. package/dist/esm/tools/web-search-tool.d.ts +2 -2
  36. package/dist/esm/tools/web-search-tool.js.map +1 -1
  37. package/package.json +6 -6
  38. package/src/adapters/chat-completions-text.ts +605 -117
  39. package/src/adapters/chat-completions-tool-converter.ts +9 -5
  40. package/src/adapters/responses-text.ts +869 -210
  41. package/src/index.ts +2 -12
  42. package/src/tools/apply-patch-tool.ts +2 -2
  43. package/src/tools/code-interpreter-tool.ts +4 -2
  44. package/src/tools/computer-use-tool.ts +2 -2
  45. package/src/tools/custom-tool.ts +2 -2
  46. package/src/tools/file-search-tool.ts +3 -3
  47. package/src/tools/function-tool.ts +2 -2
  48. package/src/tools/image-generation-tool.ts +4 -2
  49. package/src/tools/local-shell-tool.ts +4 -2
  50. package/src/tools/mcp-tool.ts +4 -2
  51. package/src/tools/shell-tool.ts +2 -2
  52. package/src/tools/web-search-preview-tool.ts +2 -2
  53. package/src/tools/web-search-tool.ts +2 -2
  54. package/dist/esm/adapters/image.d.ts +0 -32
  55. package/dist/esm/adapters/image.js +0 -89
  56. package/dist/esm/adapters/image.js.map +0 -1
  57. package/dist/esm/adapters/summarize.d.ts +0 -28
  58. package/dist/esm/adapters/summarize.js +0 -112
  59. package/dist/esm/adapters/summarize.js.map +0 -1
  60. package/dist/esm/adapters/transcription.d.ts +0 -34
  61. package/dist/esm/adapters/transcription.js +0 -131
  62. package/dist/esm/adapters/transcription.js.map +0 -1
  63. package/dist/esm/adapters/tts.d.ts +0 -26
  64. package/dist/esm/adapters/tts.js +0 -78
  65. package/dist/esm/adapters/tts.js.map +0 -1
  66. package/dist/esm/adapters/video.d.ts +0 -72
  67. package/dist/esm/adapters/video.js +0 -238
  68. package/dist/esm/adapters/video.js.map +0 -1
  69. package/dist/esm/types/config.d.ts +0 -4
  70. package/dist/esm/utils/client.d.ts +0 -3
  71. package/dist/esm/utils/client.js +0 -8
  72. package/dist/esm/utils/client.js.map +0 -1
  73. package/src/adapters/image.ts +0 -158
  74. package/src/adapters/summarize.ts +0 -174
  75. package/src/adapters/transcription.ts +0 -194
  76. package/src/adapters/tts.ts +0 -124
  77. package/src/adapters/video.ts +0 -385
  78. package/src/types/config.ts +0 -5
  79. package/src/utils/client.ts +0 -8
@@ -1,16 +1,22 @@
1
+ import { EventType } from '@tanstack/ai'
1
2
  import { BaseTextAdapter } from '@tanstack/ai/adapters'
2
3
  import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
3
4
  import { generateId, transformNullsToUndefined } from '@tanstack/ai-utils'
4
- import { createOpenAICompatibleClient } from '../utils/client'
5
5
  import { extractRequestOptions } from '../utils/request-options'
6
6
  import { makeStructuredOutputCompatible } from '../utils/schema-converter'
7
7
  import { convertToolsToResponsesFormat } from './responses-tool-converter'
8
+ import type OpenAI from 'openai'
8
9
  import type {
9
10
  StructuredOutputOptions,
10
11
  StructuredOutputResult,
11
12
  } from '@tanstack/ai/adapters'
12
- import type OpenAI_SDK from 'openai'
13
- import type { Responses } from 'openai/resources'
13
+ import type {
14
+ Response,
15
+ ResponseCreateParams,
16
+ ResponseInput,
17
+ ResponseInputContent,
18
+ ResponseStreamEvent,
19
+ } from 'openai/resources/responses/responses'
14
20
  import type {
15
21
  ContentPart,
16
22
  DefaultMessageMetadataByModality,
@@ -19,39 +25,16 @@ import type {
19
25
  StreamChunk,
20
26
  TextOptions,
21
27
  } from '@tanstack/ai'
22
- import type { OpenAICompatibleClientConfig } from '../types/config'
23
-
24
- /** Cast an event object to StreamChunk. Adapters construct events with string
25
- * literal types which are structurally compatible with the EventType enum. */
26
- const asChunk = (chunk: Record<string, unknown>) =>
27
- chunk as unknown as StreamChunk
28
28
 
29
29
  /**
30
- * OpenAI-compatible Responses API Text Adapter
31
- *
32
- * A generalized base class for providers that use the OpenAI Responses API
33
- * (`/v1/responses`). Providers like OpenAI (native), Azure OpenAI, and others
34
- * that implement the Responses API can extend this class and only need to:
35
- * - Set `baseURL` in the config
36
- * - Lock the generic type parameters to provider-specific types
37
- * - Override specific methods for quirks
38
- *
39
- * Key differences from the Chat Completions adapter:
40
- * - Uses `client.responses.create()` instead of `client.chat.completions.create()`
41
- * - Messages use `ResponseInput` format
42
- * - System prompts go in `instructions` field, not as array messages
43
- * - Streaming events are completely different (9+ event types vs simple delta chunks)
44
- * - Supports reasoning/thinking tokens via `response.reasoning_text.delta`
45
- * - Structured output uses `text.format` in the request (not `response_format`)
46
- * - Tool calls use `response.function_call_arguments.delta`
47
- * - Content parts are `input_text`, `input_image`, `input_file`
48
- *
49
- * All methods that build requests or process responses are `protected` so subclasses
50
- * can override them.
30
+ * Shared implementation of the OpenAI Responses API. Holds the stream-event
31
+ * accumulator + AG-UI lifecycle and calls the OpenAI SDK directly. Subclasses
32
+ * (today: ai-openai) construct an OpenAI client with their provider-specific
33
+ * `baseURL` / headers and pass it in.
51
34
  */
52
- export class OpenAICompatibleResponsesTextAdapter<
35
+ export abstract class OpenAIBaseResponsesTextAdapter<
53
36
  TModel extends string,
54
- TProviderOptions extends Record<string, any> = Record<string, any>,
37
+ TProviderOptions extends Record<string, unknown> = Record<string, unknown>,
55
38
  TInputModalities extends ReadonlyArray<Modality> = ReadonlyArray<Modality>,
56
39
  TMessageMetadata extends DefaultMessageMetadataByModality =
57
40
  DefaultMessageMetadataByModality,
@@ -65,17 +48,12 @@ export class OpenAICompatibleResponsesTextAdapter<
65
48
  > {
66
49
  readonly kind = 'text' as const
67
50
  readonly name: string
51
+ protected client: OpenAI
68
52
 
69
- protected client: OpenAI_SDK
70
-
71
- constructor(
72
- config: OpenAICompatibleClientConfig,
73
- model: TModel,
74
- name: string = 'openai-compatible-responses',
75
- ) {
53
+ constructor(model: TModel, name: string, client: OpenAI) {
76
54
  super({}, model)
77
55
  this.name = name
78
- this.client = createOpenAICompatibleClient(config)
56
+ this.client = client
79
57
  }
80
58
 
81
59
  async *chatStream(
@@ -87,20 +65,34 @@ export class OpenAICompatibleResponsesTextAdapter<
87
65
  // We assign our own indices as we encounter unique tool call IDs.
88
66
  const toolCallMetadata = new Map<
89
67
  string,
90
- { index: number; name: string; started: boolean }
68
+ {
69
+ index: number
70
+ name: string
71
+ started: boolean
72
+ // Set once TOOL_CALL_END has been emitted (via args.done or the
73
+ // output_item.done backfill) so the two paths don't double-emit.
74
+ ended?: boolean
75
+ // Set when args.done arrives before TOOL_CALL_START could fire
76
+ // (output_item.added lacked a name). output_item.done picks these
77
+ // up to emit the missing END.
78
+ pendingArguments?: string
79
+ }
91
80
  >()
92
- const requestParams = this.mapOptionsToRequest(options)
93
- const timestamp = Date.now()
94
81
 
95
82
  // AG-UI lifecycle tracking
96
83
  const aguiState = {
97
84
  runId: generateId(this.name),
85
+ threadId: options.threadId ?? generateId(this.name),
98
86
  messageId: generateId(this.name),
99
- timestamp,
100
87
  hasEmittedRunStarted: false,
101
88
  }
102
89
 
103
90
  try {
91
+ // mapOptionsToRequest can throw on caller-side validation failures
92
+ // (empty user content, unsupported parts, webSearchTool() rejection in
93
+ // the OpenRouter override). Keep it inside the try so those failures
94
+ // surface as RUN_ERROR events instead of iterator throws.
95
+ const requestParams = this.mapOptionsToRequest(options)
104
96
  options.logger.request(
105
97
  `activity=chat provider=${this.name} model=${this.model} messages=${options.messages.length} tools=${options.tools?.length ?? 0} stream=true`,
106
98
  { provider: this.name, model: this.model },
@@ -130,22 +122,25 @@ export class OpenAICompatibleResponsesTextAdapter<
130
122
  // Emit RUN_STARTED if not yet emitted
131
123
  if (!aguiState.hasEmittedRunStarted) {
132
124
  aguiState.hasEmittedRunStarted = true
133
- yield asChunk({
134
- type: 'RUN_STARTED',
125
+ yield {
126
+ type: EventType.RUN_STARTED,
135
127
  runId: aguiState.runId,
128
+ threadId: aguiState.threadId,
136
129
  model: options.model,
137
- timestamp,
138
- })
130
+ timestamp: Date.now(),
131
+ parentRunId: options.parentRunId,
132
+ }
139
133
  }
140
134
 
141
135
  // Emit AG-UI RUN_ERROR
142
- yield asChunk({
143
- type: 'RUN_ERROR',
144
- runId: aguiState.runId,
136
+ yield {
137
+ type: EventType.RUN_ERROR,
145
138
  model: options.model,
146
- timestamp,
139
+ timestamp: Date.now(),
140
+ message: errorPayload.message,
141
+ code: errorPayload.code,
147
142
  error: errorPayload,
148
- })
143
+ }
149
144
 
150
145
  options.logger.errors(`${this.name}.chatStream fatal`, {
151
146
  error: errorPayload,
@@ -195,10 +190,7 @@ export class OpenAICompatibleResponsesTextAdapter<
195
190
  )
196
191
  const response = await this.client.responses.create(
197
192
  {
198
- ...(cleanParams as Omit<
199
- OpenAI_SDK.Responses.ResponseCreateParams,
200
- 'stream'
201
- >),
193
+ ...(cleanParams as Omit<ResponseCreateParams, 'stream'>),
202
194
  stream: false,
203
195
  // Configure structured output via text.format
204
196
  text: {
@@ -217,9 +209,17 @@ export class OpenAICompatibleResponsesTextAdapter<
217
209
  // SDK return type to `Response`, but the explicit annotation makes
218
210
  // that contract local rather than relying on inference through the
219
211
  // overloaded `client.responses.create` signature.
220
- const rawText = this.extractTextFromResponse(
221
- response satisfies OpenAI_SDK.Responses.Response,
222
- )
212
+ const rawText = this.extractTextFromResponse(response satisfies Response)
213
+
214
+ // Fail loud on empty content rather than letting it cascade into a
215
+ // confusing "Failed to parse JSON. Content: " error — the root cause
216
+ // (the model returned no text content for the structured request) is
217
+ // then visible in logs. Mirrors the chat-completions sibling.
218
+ if (rawText.length === 0) {
219
+ throw new Error(
220
+ `${this.name}.structuredOutput: response contained no content`,
221
+ )
222
+ }
223
223
 
224
224
  // Parse the JSON response
225
225
  let parsed: unknown
@@ -231,9 +231,13 @@ export class OpenAICompatibleResponsesTextAdapter<
231
231
  )
232
232
  }
233
233
 
234
- // Transform null values to undefined to match original Zod schema expectations
235
- // Provider returns null for optional fields we made nullable in the schema
236
- const transformed = transformNullsToUndefined(parsed)
234
+ // Apply the provider-specific post-parse shaping (default: null →
235
+ // undefined to align with the original Zod schema's optional-field
236
+ // semantics; subclasses with different conventions can override
237
+ // `transformStructuredOutput`, mirroring the chat-completions base's
238
+ // hook so OpenRouter and other providers that preserve nulls in
239
+ // structured output can opt out without forking `structuredOutput`).
240
+ const transformed = this.transformStructuredOutput(parsed)
237
241
 
238
242
  return {
239
243
  data: transformed,
@@ -250,6 +254,389 @@ export class OpenAICompatibleResponsesTextAdapter<
250
254
  }
251
255
  }
252
256
 
257
+ /**
258
+ * Stream structured output via the Responses API: single request with
259
+ * `text.format: json_schema` + `stream: true`. Consumes Responses-API
260
+ * events (`response.output_text.delta`, `response.reasoning_text.delta`,
261
+ * `response.reasoning_summary_text.delta`, `response.refusal.delta`,
262
+ * `response.completed`, `response.failed`) and re-emits the standard AG-UI
263
+ * lifecycle ending with `CUSTOM 'structured-output.complete'`.
264
+ *
265
+ * Tools are stripped (structured output is mutually exclusive with tool
266
+ * calls in this path). Reasoning text is accumulated and surfaced both as
267
+ * REASONING_* lifecycle events during the stream and on the terminal
268
+ * CUSTOM event's `value.reasoning`.
269
+ */
270
+ async *structuredOutputStream(
271
+ options: StructuredOutputOptions<TProviderOptions>,
272
+ ): AsyncIterable<StreamChunk> {
273
+ const { chatOptions, outputSchema } = options
274
+ const requestParams = this.mapOptionsToRequest(chatOptions)
275
+
276
+ const jsonSchema = this.makeStructuredOutputCompatible(
277
+ outputSchema,
278
+ outputSchema.required,
279
+ )
280
+
281
+ const timestamp = Date.now()
282
+ const aguiState = {
283
+ runId: generateId(this.name),
284
+ threadId: chatOptions.threadId ?? generateId(this.name),
285
+ messageId: generateId(this.name),
286
+ timestamp,
287
+ hasEmittedRunStarted: false,
288
+ }
289
+
290
+ let accumulatedContent = ''
291
+ let accumulatedReasoning = ''
292
+ let hasEmittedTextMessageStart = false
293
+ let reasoningMessageId: string | undefined
294
+ let stepId: string | undefined
295
+ let hasClosedReasoning = false
296
+ let model: string = chatOptions.model
297
+ let usage: OpenAI.Responses.Response['usage'] | undefined
298
+
299
+ const closeReasoning = function* (this: {
300
+ name: string
301
+ }): Generator<StreamChunk> {
302
+ if (reasoningMessageId && !hasClosedReasoning) {
303
+ hasClosedReasoning = true
304
+ yield {
305
+ type: EventType.REASONING_MESSAGE_END,
306
+ messageId: reasoningMessageId,
307
+ model,
308
+ timestamp,
309
+ }
310
+ yield {
311
+ type: EventType.REASONING_END,
312
+ messageId: reasoningMessageId,
313
+ model,
314
+ timestamp,
315
+ }
316
+ if (stepId) {
317
+ yield {
318
+ type: EventType.STEP_FINISHED,
319
+ stepName: stepId,
320
+ stepId,
321
+ model,
322
+ timestamp,
323
+ content: accumulatedReasoning,
324
+ }
325
+ }
326
+ }
327
+ }.bind(this)
328
+
329
+ const openReasoning = function* (this: {
330
+ name: string
331
+ }): Generator<StreamChunk> {
332
+ if (reasoningMessageId) return
333
+ reasoningMessageId = generateId(this.name)
334
+ stepId = generateId(this.name)
335
+ yield {
336
+ type: EventType.REASONING_START,
337
+ messageId: reasoningMessageId,
338
+ model,
339
+ timestamp,
340
+ }
341
+ yield {
342
+ type: EventType.REASONING_MESSAGE_START,
343
+ messageId: reasoningMessageId,
344
+ role: 'reasoning' as const,
345
+ model,
346
+ timestamp,
347
+ }
348
+ yield {
349
+ type: EventType.STEP_STARTED,
350
+ stepName: stepId,
351
+ stepId,
352
+ model,
353
+ timestamp,
354
+ stepType: 'thinking',
355
+ }
356
+ }.bind(this)
357
+
358
+ try {
359
+ const { tools: _tools, ...cleanParams } = requestParams
360
+ void _tools
361
+
362
+ chatOptions.logger.request(
363
+ `activity=structuredOutputStream provider=${this.name} model=${this.model} messages=${chatOptions.messages.length}`,
364
+ { provider: this.name, model: this.model },
365
+ )
366
+
367
+ const stream = await this.client.responses.create(
368
+ {
369
+ ...cleanParams,
370
+ stream: true,
371
+ text: {
372
+ format: {
373
+ type: 'json_schema',
374
+ name: 'structured_output',
375
+ schema: jsonSchema,
376
+ strict: true,
377
+ },
378
+ },
379
+ },
380
+ extractRequestOptions(chatOptions.request),
381
+ )
382
+
383
+ for await (const chunk of stream) {
384
+ chatOptions.logger.provider(
385
+ `provider=${this.name} type=${chunk.type}`,
386
+ { provider: this.name, type: chunk.type },
387
+ )
388
+
389
+ if (!aguiState.hasEmittedRunStarted) {
390
+ aguiState.hasEmittedRunStarted = true
391
+ yield {
392
+ type: EventType.RUN_STARTED,
393
+ runId: aguiState.runId,
394
+ threadId: aguiState.threadId,
395
+ model,
396
+ timestamp,
397
+ parentRunId: chatOptions.parentRunId,
398
+ }
399
+ }
400
+
401
+ if (
402
+ chunk.type === 'response.created' ||
403
+ chunk.type === 'response.in_progress'
404
+ ) {
405
+ const responseModel = (chunk as { response?: { model?: string } })
406
+ .response?.model
407
+ if (responseModel) model = responseModel
408
+ continue
409
+ }
410
+
411
+ if (chunk.type === 'response.refusal.delta') {
412
+ const delta =
413
+ typeof (chunk as { delta?: unknown }).delta === 'string'
414
+ ? (chunk as { delta: string }).delta
415
+ : ''
416
+ yield {
417
+ type: EventType.RUN_ERROR,
418
+ runId: aguiState.runId,
419
+ model,
420
+ timestamp,
421
+ message: `Model refused: ${delta}`,
422
+ code: 'refusal',
423
+ error: { message: `Model refused: ${delta}`, code: 'refusal' },
424
+ }
425
+ return
426
+ }
427
+
428
+ if (
429
+ chunk.type === 'response.reasoning_text.delta' ||
430
+ chunk.type === 'response.reasoning_summary_text.delta'
431
+ ) {
432
+ const raw = (chunk as { delta?: unknown }).delta
433
+ const reasoningDelta = Array.isArray(raw)
434
+ ? raw.join('')
435
+ : typeof raw === 'string'
436
+ ? raw
437
+ : ''
438
+ if (!reasoningDelta) continue
439
+ yield* openReasoning()
440
+ // openReasoning() guarantees reasoningMessageId is set on first call;
441
+ // TS can't see through the generator side-effect.
442
+ const messageId = reasoningMessageId!
443
+ accumulatedReasoning += reasoningDelta
444
+ yield {
445
+ type: EventType.REASONING_MESSAGE_CONTENT,
446
+ messageId,
447
+ delta: reasoningDelta,
448
+ model,
449
+ timestamp,
450
+ }
451
+ continue
452
+ }
453
+
454
+ if (chunk.type === 'response.output_text.delta') {
455
+ const raw = (chunk as { delta?: unknown }).delta
456
+ const textDelta = Array.isArray(raw)
457
+ ? raw.join('')
458
+ : typeof raw === 'string'
459
+ ? raw
460
+ : ''
461
+ if (!textDelta) continue
462
+
463
+ yield* closeReasoning()
464
+
465
+ if (!hasEmittedTextMessageStart) {
466
+ hasEmittedTextMessageStart = true
467
+ yield {
468
+ type: EventType.TEXT_MESSAGE_START,
469
+ messageId: aguiState.messageId,
470
+ model,
471
+ timestamp,
472
+ role: 'assistant',
473
+ }
474
+ }
475
+ accumulatedContent += textDelta
476
+ yield {
477
+ type: EventType.TEXT_MESSAGE_CONTENT,
478
+ messageId: aguiState.messageId,
479
+ model,
480
+ timestamp,
481
+ delta: textDelta,
482
+ content: accumulatedContent,
483
+ }
484
+ continue
485
+ }
486
+
487
+ if (chunk.type === 'response.completed') {
488
+ const response = chunk.response
489
+ if (response.usage) usage = response.usage
490
+ if (response.model) model = response.model
491
+ continue
492
+ }
493
+
494
+ if (chunk.type === 'response.failed') {
495
+ const response = (
496
+ chunk as {
497
+ response?: { error?: { message?: string; code?: string } }
498
+ }
499
+ ).response
500
+ const message =
501
+ response?.error?.message || 'Responses API stream failed'
502
+ yield {
503
+ type: EventType.RUN_ERROR,
504
+ runId: aguiState.runId,
505
+ model,
506
+ timestamp,
507
+ message,
508
+ code: response?.error?.code,
509
+ error: { message, code: response?.error?.code },
510
+ }
511
+ return
512
+ }
513
+ }
514
+
515
+ yield* closeReasoning()
516
+
517
+ if (hasEmittedTextMessageStart) {
518
+ yield {
519
+ type: EventType.TEXT_MESSAGE_END,
520
+ messageId: aguiState.messageId,
521
+ model,
522
+ timestamp,
523
+ }
524
+ }
525
+
526
+ if (accumulatedContent.length === 0) {
527
+ yield {
528
+ type: EventType.RUN_ERROR,
529
+ runId: aguiState.runId,
530
+ model,
531
+ timestamp,
532
+ message: `${this.name}.structuredOutputStream: response contained no content`,
533
+ code: 'empty-response',
534
+ error: {
535
+ message: `${this.name}.structuredOutputStream: response contained no content`,
536
+ code: 'empty-response',
537
+ },
538
+ }
539
+ return
540
+ }
541
+
542
+ let parsed: unknown
543
+ try {
544
+ parsed = JSON.parse(accumulatedContent)
545
+ } catch {
546
+ yield {
547
+ type: EventType.RUN_ERROR,
548
+ runId: aguiState.runId,
549
+ model,
550
+ timestamp,
551
+ message: `Failed to parse structured output as JSON. Content: ${accumulatedContent.slice(0, 200)}${accumulatedContent.length > 200 ? '...' : ''}`,
552
+ code: 'parse-error',
553
+ error: {
554
+ message: 'Failed to parse structured output as JSON',
555
+ code: 'parse-error',
556
+ },
557
+ }
558
+ return
559
+ }
560
+
561
+ const transformed = transformNullsToUndefined(parsed)
562
+
563
+ yield {
564
+ type: EventType.CUSTOM,
565
+ name: 'structured-output.complete',
566
+ value: {
567
+ object: transformed,
568
+ raw: accumulatedContent,
569
+ ...(accumulatedReasoning ? { reasoning: accumulatedReasoning } : {}),
570
+ },
571
+ model,
572
+ timestamp,
573
+ }
574
+
575
+ yield {
576
+ type: EventType.RUN_FINISHED,
577
+ runId: aguiState.runId,
578
+ threadId: aguiState.threadId,
579
+ model,
580
+ timestamp,
581
+ finishReason: 'stop',
582
+ ...(usage && {
583
+ usage: {
584
+ promptTokens: usage.input_tokens,
585
+ completionTokens: usage.output_tokens,
586
+ totalTokens: usage.total_tokens,
587
+ },
588
+ }),
589
+ }
590
+ } catch (error: unknown) {
591
+ if (!aguiState.hasEmittedRunStarted) {
592
+ aguiState.hasEmittedRunStarted = true
593
+ yield {
594
+ type: EventType.RUN_STARTED,
595
+ runId: aguiState.runId,
596
+ threadId: aguiState.threadId,
597
+ model,
598
+ timestamp,
599
+ parentRunId: chatOptions.parentRunId,
600
+ }
601
+ }
602
+
603
+ const isAbort = this.isAbortError(error)
604
+ const errorPayload = toRunErrorPayload(
605
+ error,
606
+ `${this.name}.structuredOutputStream failed`,
607
+ )
608
+
609
+ yield {
610
+ type: EventType.RUN_ERROR,
611
+ runId: aguiState.runId,
612
+ model,
613
+ timestamp,
614
+ message: errorPayload.message,
615
+ code: isAbort ? 'aborted' : errorPayload.code,
616
+ error: { ...errorPayload, ...(isAbort && { code: 'aborted' }) },
617
+ }
618
+
619
+ chatOptions.logger.errors(`${this.name}.structuredOutputStream fatal`, {
620
+ error: errorPayload,
621
+ source: `${this.name}.structuredOutputStream`,
622
+ })
623
+ }
624
+ }
625
+
626
+ /**
627
+ * Cross-SDK abort detection for `structuredOutputStream`. Mirrors the
628
+ * Chat Completions base; subclasses with proprietary error types override.
629
+ */
630
+ protected isAbortError(error: unknown): boolean {
631
+ if (!error || typeof error !== 'object') return false
632
+ const e = error as { name?: unknown; code?: unknown }
633
+ return (
634
+ e.name === 'APIUserAbortError' ||
635
+ e.name === 'AbortError' ||
636
+ e.code === 'ERR_CANCELED'
637
+ )
638
+ }
639
+
253
640
  /**
254
641
  * Applies provider-specific transformations for structured output compatibility.
255
642
  * Override this in subclasses to handle provider-specific quirks.
@@ -261,26 +648,49 @@ export class OpenAICompatibleResponsesTextAdapter<
261
648
  return makeStructuredOutputCompatible(schema, originalRequired)
262
649
  }
263
650
 
651
+ /**
652
+ * Final shaping pass applied to parsed structured-output JSON before it is
653
+ * returned to the caller. Default converts `null` values to `undefined` so
654
+ * the result aligns with the original Zod schema's optional-field
655
+ * semantics. Subclasses with different conventions (OpenRouter historically
656
+ * preserves nulls) can override — mirrors the chat-completions base's hook
657
+ * so a subclass that opts out of null-stripping doesn't have to fork the
658
+ * whole `structuredOutput` method.
659
+ */
660
+ protected transformStructuredOutput(parsed: unknown): unknown {
661
+ return transformNullsToUndefined(parsed)
662
+ }
663
+
264
664
  /**
265
665
  * Extract text content from a non-streaming Responses API response.
266
666
  * Override this in subclasses for provider-specific response shapes.
267
667
  */
268
- protected extractTextFromResponse(
269
- response: OpenAI_SDK.Responses.Response,
270
- ): string {
668
+ protected extractTextFromResponse(response: Response): string {
271
669
  let textContent = ''
272
670
  let refusal: string | undefined
671
+ let sawMessageItem = false
672
+ const observedItemTypes = new Set<string>()
273
673
 
274
674
  for (const item of response.output) {
675
+ observedItemTypes.add(item.type)
275
676
  if (item.type === 'message') {
677
+ sawMessageItem = true
276
678
  for (const part of item.content) {
277
- if (part.type === 'output_text') {
278
- textContent += part.text
679
+ // Cast off the discriminated union before the type discrimination
680
+ // so future SDK variants (e.g. `output_audio`, `output_image`) hit
681
+ // the explicit error path rather than being misreported as refusals
682
+ // when they get added to the union. Mirrors the streaming side's
683
+ // handleContentPart.
684
+ const partType = (part as { type: string }).type
685
+ if (partType === 'output_text') {
686
+ textContent += (part as { text?: string }).text ?? ''
687
+ } else if (partType === 'refusal') {
688
+ const refusalText = (part as { refusal?: string }).refusal
689
+ refusal = refusalText || refusal || 'Refused without explanation'
279
690
  } else {
280
- // The Responses SDK currently models message content as
281
- // `output_text | refusal`, so the only non-text branch is a
282
- // refusal. Capture it so we can surface a distinct error below.
283
- refusal = part.refusal || refusal || 'Refused without explanation'
691
+ throw new Error(
692
+ `${this.name}.extractTextFromResponse: unsupported message content part type "${partType}"`,
693
+ )
284
694
  }
285
695
  }
286
696
  }
@@ -295,6 +705,16 @@ export class OpenAICompatibleResponsesTextAdapter<
295
705
  throw err
296
706
  }
297
707
 
708
+ // Response had items but none carried message text (e.g. only
709
+ // function_call or reasoning items). Surface that explicitly so a
710
+ // downstream structured-output caller doesn't see a misleading
711
+ // "Failed to parse JSON. Content: " from an empty string.
712
+ if (!textContent && response.output.length > 0 && !sawMessageItem) {
713
+ throw new Error(
714
+ `${this.name}.extractTextFromResponse: response.output contained items of type(s) [${[...observedItemTypes].sort().join(', ')}] but no message text — the model returned a non-text response`,
715
+ )
716
+ }
717
+
298
718
  return textContent
299
719
  }
300
720
 
@@ -314,22 +734,27 @@ export class OpenAICompatibleResponsesTextAdapter<
314
734
  * - error
315
735
  */
316
736
  protected async *processStreamChunks(
317
- stream: AsyncIterable<OpenAI_SDK.Responses.ResponseStreamEvent>,
737
+ stream: AsyncIterable<ResponseStreamEvent>,
318
738
  toolCallMetadata: Map<
319
739
  string,
320
- { index: number; name: string; started: boolean }
740
+ {
741
+ index: number
742
+ name: string
743
+ started: boolean
744
+ ended?: boolean
745
+ pendingArguments?: string
746
+ }
321
747
  >,
322
748
  options: TextOptions<TProviderOptions>,
323
749
  aguiState: {
324
750
  runId: string
751
+ threadId: string
325
752
  messageId: string
326
- timestamp: number
327
753
  hasEmittedRunStarted: boolean
328
754
  },
329
755
  ): AsyncIterable<StreamChunk> {
330
756
  let accumulatedContent = ''
331
757
  let accumulatedReasoning = ''
332
- const timestamp = aguiState.timestamp
333
758
 
334
759
  // Track if we've been streaming deltas to avoid duplicating content from done events
335
760
  let hasStreamedContentDeltas = false
@@ -357,12 +782,14 @@ export class OpenAICompatibleResponsesTextAdapter<
357
782
  // Emit RUN_STARTED on first chunk
358
783
  if (!aguiState.hasEmittedRunStarted) {
359
784
  aguiState.hasEmittedRunStarted = true
360
- yield asChunk({
361
- type: 'RUN_STARTED',
785
+ yield {
786
+ type: EventType.RUN_STARTED,
362
787
  runId: aguiState.runId,
788
+ threadId: aguiState.threadId,
363
789
  model: model || options.model,
364
- timestamp,
365
- })
790
+ timestamp: Date.now(),
791
+ parentRunId: options.parentRunId,
792
+ }
366
793
  }
367
794
 
368
795
  const handleContentPart = (contentPart: {
@@ -372,14 +799,14 @@ export class OpenAICompatibleResponsesTextAdapter<
372
799
  }): StreamChunk => {
373
800
  if (contentPart.type === 'output_text') {
374
801
  accumulatedContent += contentPart.text || ''
375
- return asChunk({
376
- type: 'TEXT_MESSAGE_CONTENT',
802
+ return {
803
+ type: EventType.TEXT_MESSAGE_CONTENT,
377
804
  messageId: aguiState.messageId,
378
805
  model: model || options.model,
379
- timestamp,
806
+ timestamp: Date.now(),
380
807
  delta: contentPart.text || '',
381
808
  content: accumulatedContent,
382
- })
809
+ }
383
810
  }
384
811
 
385
812
  if (contentPart.type === 'reasoning_text') {
@@ -391,14 +818,15 @@ export class OpenAICompatibleResponsesTextAdapter<
391
818
  if (!stepId) {
392
819
  stepId = generateId(this.name)
393
820
  }
394
- return asChunk({
395
- type: 'STEP_FINISHED',
821
+ return {
822
+ type: EventType.STEP_FINISHED,
823
+ stepName: stepId,
396
824
  stepId,
397
825
  model: model || options.model,
398
- timestamp,
826
+ timestamp: Date.now(),
399
827
  delta: contentPart.text || '',
400
828
  content: accumulatedReasoning,
401
- })
829
+ }
402
830
  }
403
831
  // Either a real refusal or an unknown content_part type. Surface
404
832
  // the part type in the error so unknown parts are debuggable
@@ -407,16 +835,15 @@ export class OpenAICompatibleResponsesTextAdapter<
407
835
  const message = isRefusal
408
836
  ? contentPart.refusal || 'Refused without explanation'
409
837
  : `Unsupported response content_part type: ${contentPart.type}`
410
- return asChunk({
411
- type: 'RUN_ERROR',
412
- runId: aguiState.runId,
838
+ const code = isRefusal ? 'refusal' : contentPart.type
839
+ return {
840
+ type: EventType.RUN_ERROR,
413
841
  model: model || options.model,
414
- timestamp,
415
- error: {
416
- message,
417
- code: isRefusal ? 'refusal' : contentPart.type,
418
- },
419
- })
842
+ timestamp: Date.now(),
843
+ message,
844
+ code,
845
+ error: { message, code },
846
+ }
420
847
  }
421
848
 
422
849
  // Capture model metadata from any of these events (created starts
@@ -451,12 +878,12 @@ export class OpenAICompatibleResponsesTextAdapter<
451
878
  chunk.type === 'response.incomplete'
452
879
  ) {
453
880
  if (hasEmittedTextMessageStart) {
454
- yield asChunk({
455
- type: 'TEXT_MESSAGE_END',
881
+ yield {
882
+ type: EventType.TEXT_MESSAGE_END,
456
883
  messageId: aguiState.messageId,
457
884
  model: chunk.response.model,
458
- timestamp,
459
- })
885
+ timestamp: Date.now(),
886
+ }
460
887
  hasEmittedTextMessageStart = false
461
888
  }
462
889
  // Coalesce error + incomplete_details into a single RUN_ERROR
@@ -469,23 +896,25 @@ export class OpenAICompatibleResponsesTextAdapter<
469
896
  ? 'Response failed'
470
897
  : 'Response ended incomplete')
471
898
  const errorCode =
472
- chunk.response.error?.code ||
473
- (chunk.response.incomplete_details ? 'incomplete' : undefined)
899
+ chunk.response.error?.code ??
900
+ (chunk.response.incomplete_details ? 'incomplete' : undefined) ??
901
+ undefined
474
902
  // Always emit RUN_ERROR for terminal failure events, even when the
475
903
  // upstream omitted both `error` and `incomplete_details`. Skipping
476
904
  // emission on a `response.incomplete` with no detail would let the
477
905
  // post-loop synthetic block silently coerce the run to a clean
478
906
  // `RUN_FINISHED { finishReason: 'stop' }` — masking the failure.
479
- yield asChunk({
480
- type: 'RUN_ERROR',
481
- runId: aguiState.runId,
907
+ yield {
908
+ type: EventType.RUN_ERROR,
482
909
  model: chunk.response.model,
483
- timestamp,
910
+ timestamp: Date.now(),
911
+ message: errorMessage,
912
+ ...(errorCode !== undefined && { code: errorCode }),
484
913
  error: {
485
914
  message: errorMessage,
486
915
  ...(errorCode !== undefined && { code: errorCode }),
487
916
  },
488
- })
917
+ }
489
918
  // RUN_ERROR is the terminal event for this run; stop processing
490
919
  // any further chunks the iterator might still deliver.
491
920
  runFinishedEmitted = true
@@ -506,25 +935,25 @@ export class OpenAICompatibleResponsesTextAdapter<
506
935
  // Emit TEXT_MESSAGE_START on first text content
507
936
  if (!hasEmittedTextMessageStart) {
508
937
  hasEmittedTextMessageStart = true
509
- yield asChunk({
510
- type: 'TEXT_MESSAGE_START',
938
+ yield {
939
+ type: EventType.TEXT_MESSAGE_START,
511
940
  messageId: aguiState.messageId,
512
941
  model: model || options.model,
513
- timestamp,
942
+ timestamp: Date.now(),
514
943
  role: 'assistant',
515
- })
944
+ }
516
945
  }
517
946
 
518
947
  accumulatedContent += textDelta
519
948
  hasStreamedContentDeltas = true
520
- yield asChunk({
521
- type: 'TEXT_MESSAGE_CONTENT',
949
+ yield {
950
+ type: EventType.TEXT_MESSAGE_CONTENT,
522
951
  messageId: aguiState.messageId,
523
952
  model: model || options.model,
524
- timestamp,
953
+ timestamp: Date.now(),
525
954
  delta: textDelta,
526
955
  content: accumulatedContent,
527
- })
956
+ }
528
957
  }
529
958
  }
530
959
 
@@ -543,25 +972,28 @@ export class OpenAICompatibleResponsesTextAdapter<
543
972
  if (!hasEmittedStepStarted) {
544
973
  hasEmittedStepStarted = true
545
974
  stepId = generateId(this.name)
546
- yield asChunk({
547
- type: 'STEP_STARTED',
975
+ yield {
976
+ type: EventType.STEP_STARTED,
977
+ stepName: stepId,
548
978
  stepId,
549
979
  model: model || options.model,
550
- timestamp,
980
+ timestamp: Date.now(),
551
981
  stepType: 'thinking',
552
- })
982
+ }
553
983
  }
554
984
 
555
985
  accumulatedReasoning += reasoningDelta
556
986
  hasStreamedReasoningDeltas = true
557
- yield asChunk({
558
- type: 'STEP_FINISHED',
559
- stepId: stepId || generateId(this.name),
987
+ const fallbackStepId = stepId || generateId(this.name)
988
+ yield {
989
+ type: EventType.STEP_FINISHED,
990
+ stepName: fallbackStepId,
991
+ stepId: fallbackStepId,
560
992
  model: model || options.model,
561
- timestamp,
993
+ timestamp: Date.now(),
562
994
  delta: reasoningDelta,
563
995
  content: accumulatedReasoning,
564
- })
996
+ }
565
997
  }
566
998
  }
567
999
 
@@ -579,25 +1011,28 @@ export class OpenAICompatibleResponsesTextAdapter<
579
1011
  if (!hasEmittedStepStarted) {
580
1012
  hasEmittedStepStarted = true
581
1013
  stepId = generateId(this.name)
582
- yield asChunk({
583
- type: 'STEP_STARTED',
1014
+ yield {
1015
+ type: EventType.STEP_STARTED,
1016
+ stepName: stepId,
584
1017
  stepId,
585
1018
  model: model || options.model,
586
- timestamp,
1019
+ timestamp: Date.now(),
587
1020
  stepType: 'thinking',
588
- })
1021
+ }
589
1022
  }
590
1023
 
591
1024
  accumulatedReasoning += summaryDelta
592
1025
  hasStreamedReasoningDeltas = true
593
- yield asChunk({
594
- type: 'STEP_FINISHED',
595
- stepId: stepId || generateId(this.name),
1026
+ const fallbackStepId = stepId || generateId(this.name)
1027
+ yield {
1028
+ type: EventType.STEP_FINISHED,
1029
+ stepName: fallbackStepId,
1030
+ stepId: fallbackStepId,
596
1031
  model: model || options.model,
597
- timestamp,
1032
+ timestamp: Date.now(),
598
1033
  delta: summaryDelta,
599
1034
  content: accumulatedReasoning,
600
- })
1035
+ }
601
1036
  }
602
1037
  }
603
1038
 
@@ -610,25 +1045,26 @@ export class OpenAICompatibleResponsesTextAdapter<
610
1045
  !hasEmittedTextMessageStart
611
1046
  ) {
612
1047
  hasEmittedTextMessageStart = true
613
- yield asChunk({
614
- type: 'TEXT_MESSAGE_START',
1048
+ yield {
1049
+ type: EventType.TEXT_MESSAGE_START,
615
1050
  messageId: aguiState.messageId,
616
1051
  model: model || options.model,
617
- timestamp,
1052
+ timestamp: Date.now(),
618
1053
  role: 'assistant',
619
- })
1054
+ }
620
1055
  }
621
1056
  // Emit STEP_STARTED if this is reasoning content
622
1057
  if (contentPart.type === 'reasoning_text' && !hasEmittedStepStarted) {
623
1058
  hasEmittedStepStarted = true
624
1059
  stepId = generateId(this.name)
625
- yield asChunk({
626
- type: 'STEP_STARTED',
1060
+ yield {
1061
+ type: EventType.STEP_STARTED,
1062
+ stepName: stepId,
627
1063
  stepId,
628
1064
  model: model || options.model,
629
- timestamp,
1065
+ timestamp: Date.now(),
630
1066
  stepType: 'thinking',
631
- })
1067
+ }
632
1068
  }
633
1069
  // Mark whichever stream we just emitted into so a subsequent
634
1070
  // `content_part.done` doesn't duplicate the same text. Without
@@ -668,6 +1104,40 @@ export class OpenAICompatibleResponsesTextAdapter<
668
1104
  continue
669
1105
  }
670
1106
 
1107
+ // Upstreams that emit `content_part.done` without any preceding
1108
+ // deltas (or `content_part.added`) still need a START event before
1109
+ // CONTENT — otherwise consumers tracking start/end pairs see content
1110
+ // without a start and never see an end. Emit the lifecycle opener
1111
+ // for whichever stream this content_part belongs to before yielding
1112
+ // the CONTENT chunk; the post-loop block emits the matching END.
1113
+ if (
1114
+ contentPart.type === 'output_text' &&
1115
+ !hasEmittedTextMessageStart
1116
+ ) {
1117
+ hasEmittedTextMessageStart = true
1118
+ yield {
1119
+ type: EventType.TEXT_MESSAGE_START,
1120
+ messageId: aguiState.messageId,
1121
+ model: model || options.model,
1122
+ timestamp: Date.now(),
1123
+ role: 'assistant',
1124
+ }
1125
+ } else if (
1126
+ contentPart.type === 'reasoning_text' &&
1127
+ !hasEmittedStepStarted
1128
+ ) {
1129
+ hasEmittedStepStarted = true
1130
+ stepId = generateId(this.name)
1131
+ yield {
1132
+ type: EventType.STEP_STARTED,
1133
+ stepName: stepId,
1134
+ stepId,
1135
+ model: model || options.model,
1136
+ timestamp: Date.now(),
1137
+ stepType: 'thinking',
1138
+ }
1139
+ }
1140
+
671
1141
  // Only emit if we haven't been streaming deltas (e.g., for non-streaming responses)
672
1142
  const doneChunk = handleContentPart(contentPart)
673
1143
  yield doneChunk
@@ -682,27 +1152,35 @@ export class OpenAICompatibleResponsesTextAdapter<
682
1152
  const item = chunk.item
683
1153
  if (item.type === 'function_call' && item.id) {
684
1154
  const existing = toolCallMetadata.get(item.id)
685
- // Only emit TOOL_CALL_START on the FIRST output_item.added for
686
- // an item id. A duplicate emission (which can happen on retried
687
- // streams or replay) would violate AG-UI's start-once contract.
688
- if (!existing?.started) {
689
- if (!existing) {
690
- toolCallMetadata.set(item.id, {
691
- index: chunk.output_index,
692
- name: item.name || '',
693
- started: false,
694
- })
695
- }
696
- yield asChunk({
697
- type: 'TOOL_CALL_START',
1155
+ // Track the item as soon as we see it so subsequent arg deltas
1156
+ // aren't logged as orphans, but only emit TOOL_CALL_START when
1157
+ // both id AND name are populated. Emitting START with an empty
1158
+ // name would propagate into TOOL_CALL_END (which reads the same
1159
+ // metadata) and route the tool call to whatever name happens to
1160
+ // match `''` downstream — a silent misroute.
1161
+ if (!existing) {
1162
+ toolCallMetadata.set(item.id, {
1163
+ index: chunk.output_index,
1164
+ name: item.name || '',
1165
+ started: false,
1166
+ })
1167
+ } else if (!existing.name && item.name) {
1168
+ // A later output_item.added for the same id finally carries
1169
+ // the name. Update so the gated emission below can fire.
1170
+ existing.name = item.name
1171
+ }
1172
+ const metadata = toolCallMetadata.get(item.id)!
1173
+ if (!metadata.started && metadata.name) {
1174
+ yield {
1175
+ type: EventType.TOOL_CALL_START,
698
1176
  toolCallId: item.id,
699
- toolCallName: item.name || '',
700
- toolName: item.name || '',
1177
+ toolCallName: metadata.name,
1178
+ toolName: metadata.name,
701
1179
  model: model || options.model,
702
- timestamp,
1180
+ timestamp: Date.now(),
703
1181
  index: chunk.output_index,
704
- })
705
- toolCallMetadata.get(item.id)!.started = true
1182
+ }
1183
+ metadata.started = true
706
1184
  }
707
1185
  }
708
1186
  }
@@ -735,13 +1213,13 @@ export class OpenAICompatibleResponsesTextAdapter<
735
1213
  )
736
1214
  continue
737
1215
  }
738
- yield asChunk({
739
- type: 'TOOL_CALL_ARGS',
1216
+ yield {
1217
+ type: EventType.TOOL_CALL_ARGS,
740
1218
  toolCallId: chunk.item_id,
741
1219
  model: model || options.model,
742
- timestamp,
1220
+ timestamp: Date.now(),
743
1221
  delta: chunk.delta,
744
- })
1222
+ }
745
1223
  }
746
1224
 
747
1225
  if (chunk.type === 'response.function_call_arguments.done') {
@@ -749,13 +1227,19 @@ export class OpenAICompatibleResponsesTextAdapter<
749
1227
 
750
1228
  // Get the function name from metadata (captured in output_item.added)
751
1229
  const metadata = toolCallMetadata.get(item_id)
752
- // Skip TOOL_CALL_END for items whose start was never emitted (no
753
- // matching `output_item.added`). Emitting END without START would
754
- // produce an unbalanced AG-UI lifecycle event downstream consumers
755
- // can't pair.
1230
+ // If the matching START was never emitted (the upstream sent an
1231
+ // `output_item.added` without a name and no later event has filled
1232
+ // it in yet), defer END until `output_item.done` or
1233
+ // `response.completed` can backfill the name. We stash the raw
1234
+ // arguments so the late emission has them. Emitting END without
1235
+ // START would produce an unbalanced AG-UI lifecycle event
1236
+ // downstream consumers can't pair.
756
1237
  if (!metadata?.started) {
1238
+ if (metadata) {
1239
+ metadata.pendingArguments = chunk.arguments
1240
+ }
757
1241
  options.logger.errors(
758
- `${this.name}.processStreamChunks orphan function_call_arguments.done`,
1242
+ `${this.name}.processStreamChunks deferring function_call_arguments.done — TOOL_CALL_START not yet emitted (waiting for name)`,
759
1243
  {
760
1244
  source: `${this.name}.processStreamChunks`,
761
1245
  toolCallId: item_id,
@@ -764,7 +1248,12 @@ export class OpenAICompatibleResponsesTextAdapter<
764
1248
  )
765
1249
  continue
766
1250
  }
1251
+ // The output_item.done backstop may have already emitted END (when
1252
+ // it arrived before args.done with a populated item.arguments).
1253
+ // Skip so we never produce a duplicate close for the same id.
1254
+ if (metadata.ended) continue
767
1255
  const name = metadata.name || ''
1256
+ metadata.ended = true
768
1257
 
769
1258
  // Parse arguments. Surface parse failures via the logger so a
770
1259
  // model emitting malformed JSON is debuggable instead of silently
@@ -792,26 +1281,177 @@ export class OpenAICompatibleResponsesTextAdapter<
792
1281
  }
793
1282
  }
794
1283
 
795
- yield asChunk({
796
- type: 'TOOL_CALL_END',
1284
+ yield {
1285
+ type: EventType.TOOL_CALL_END,
797
1286
  toolCallId: item_id,
798
1287
  toolCallName: name,
799
1288
  toolName: name,
800
1289
  model: model || options.model,
801
- timestamp,
1290
+ timestamp: Date.now(),
802
1291
  input: parsedInput,
803
- })
1292
+ }
1293
+ }
1294
+
1295
+ // `output_item.done` is the last point at which a function_call's
1296
+ // name is guaranteed to be on the wire — it carries the fully-formed
1297
+ // ResponseFunctionToolCall. Use it as a backstop to recover any
1298
+ // tool call whose name was missing from `output_item.added` (and
1299
+ // whose START + END therefore never fired).
1300
+ if (chunk.type === 'response.output_item.done') {
1301
+ const item = chunk.item
1302
+ if (item.type === 'function_call' && item.id) {
1303
+ const metadata = toolCallMetadata.get(item.id) ?? {
1304
+ index: chunk.output_index,
1305
+ name: item.name || '',
1306
+ started: false,
1307
+ }
1308
+ if (!toolCallMetadata.has(item.id)) {
1309
+ toolCallMetadata.set(item.id, metadata)
1310
+ } else if (!metadata.name && item.name) {
1311
+ metadata.name = item.name
1312
+ }
1313
+ // Emit gated START if we now have a name and never started.
1314
+ if (!metadata.started && metadata.name) {
1315
+ yield {
1316
+ type: EventType.TOOL_CALL_START,
1317
+ toolCallId: item.id,
1318
+ toolCallName: metadata.name,
1319
+ toolName: metadata.name,
1320
+ model: model || options.model,
1321
+ timestamp: Date.now(),
1322
+ index: metadata.index,
1323
+ }
1324
+ metadata.started = true
1325
+ }
1326
+ // Emit END if we have args (either from a previously-deferred
1327
+ // args.done OR from item.arguments) and haven't already ended.
1328
+ const rawArgs =
1329
+ typeof item.arguments === 'string' && item.arguments.length > 0
1330
+ ? item.arguments
1331
+ : metadata.pendingArguments
1332
+ if (metadata.started && !metadata.ended && rawArgs !== undefined) {
1333
+ const name = metadata.name || ''
1334
+ let parsedInput: unknown = {}
1335
+ if (rawArgs) {
1336
+ try {
1337
+ const parsed = JSON.parse(rawArgs)
1338
+ parsedInput =
1339
+ parsed && typeof parsed === 'object' ? parsed : {}
1340
+ } catch (parseError) {
1341
+ options.logger.errors(
1342
+ `${this.name}.processStreamChunks tool-args JSON parse failed (output_item.done backfill)`,
1343
+ {
1344
+ error: toRunErrorPayload(
1345
+ parseError,
1346
+ `tool ${name} (${item.id}) returned malformed JSON arguments`,
1347
+ ),
1348
+ source: `${this.name}.processStreamChunks`,
1349
+ toolCallId: item.id,
1350
+ toolName: name,
1351
+ rawArguments: rawArgs,
1352
+ },
1353
+ )
1354
+ parsedInput = {}
1355
+ }
1356
+ }
1357
+ yield {
1358
+ type: EventType.TOOL_CALL_END,
1359
+ toolCallId: item.id,
1360
+ toolCallName: name,
1361
+ toolName: name,
1362
+ model: model || options.model,
1363
+ timestamp: Date.now(),
1364
+ input: parsedInput,
1365
+ }
1366
+ metadata.ended = true
1367
+ metadata.pendingArguments = undefined
1368
+ }
1369
+ }
804
1370
  }
805
1371
 
806
1372
  if (chunk.type === 'response.completed') {
1373
+ // Final backstop for function_call lifecycle: if a function_call
1374
+ // appears in `response.output[]` but was never matched by an
1375
+ // output_item.added/done with a name, recover the missing START
1376
+ // (and END if args were pending). Without this, a tool call could
1377
+ // be silently dropped from the AG-UI stream while `hasFunctionCalls`
1378
+ // below still routes the run's finishReason to 'tool_calls' —
1379
+ // leaving consumers waiting for tool results they never saw start.
1380
+ for (const item of chunk.response.output) {
1381
+ if (item.type !== 'function_call' || !item.id) continue
1382
+ const metadata = toolCallMetadata.get(item.id) ?? {
1383
+ index: 0,
1384
+ name: item.name || '',
1385
+ started: false,
1386
+ }
1387
+ if (!toolCallMetadata.has(item.id)) {
1388
+ toolCallMetadata.set(item.id, metadata)
1389
+ } else if (!metadata.name && item.name) {
1390
+ metadata.name = item.name
1391
+ }
1392
+ if (!metadata.started && metadata.name) {
1393
+ yield {
1394
+ type: EventType.TOOL_CALL_START,
1395
+ toolCallId: item.id,
1396
+ toolCallName: metadata.name,
1397
+ toolName: metadata.name,
1398
+ model: model || options.model,
1399
+ timestamp: Date.now(),
1400
+ index: metadata.index,
1401
+ }
1402
+ metadata.started = true
1403
+ }
1404
+ const rawArgs =
1405
+ typeof item.arguments === 'string' && item.arguments.length > 0
1406
+ ? item.arguments
1407
+ : metadata.pendingArguments
1408
+ if (metadata.started && !metadata.ended) {
1409
+ const name = metadata.name || ''
1410
+ let parsedInput: unknown = {}
1411
+ if (rawArgs) {
1412
+ try {
1413
+ const parsed = JSON.parse(rawArgs)
1414
+ parsedInput =
1415
+ parsed && typeof parsed === 'object' ? parsed : {}
1416
+ } catch (parseError) {
1417
+ options.logger.errors(
1418
+ `${this.name}.processStreamChunks tool-args JSON parse failed (response.completed backfill)`,
1419
+ {
1420
+ error: toRunErrorPayload(
1421
+ parseError,
1422
+ `tool ${name} (${item.id}) returned malformed JSON arguments`,
1423
+ ),
1424
+ source: `${this.name}.processStreamChunks`,
1425
+ toolCallId: item.id,
1426
+ toolName: name,
1427
+ rawArguments: rawArgs,
1428
+ },
1429
+ )
1430
+ parsedInput = {}
1431
+ }
1432
+ }
1433
+ yield {
1434
+ type: EventType.TOOL_CALL_END,
1435
+ toolCallId: item.id,
1436
+ toolCallName: name,
1437
+ toolName: name,
1438
+ model: model || options.model,
1439
+ timestamp: Date.now(),
1440
+ input: parsedInput,
1441
+ }
1442
+ metadata.ended = true
1443
+ metadata.pendingArguments = undefined
1444
+ }
1445
+ }
1446
+
807
1447
  // Emit TEXT_MESSAGE_END if we had text content
808
1448
  if (hasEmittedTextMessageStart) {
809
- yield asChunk({
810
- type: 'TEXT_MESSAGE_END',
1449
+ yield {
1450
+ type: EventType.TEXT_MESSAGE_END,
811
1451
  messageId: aguiState.messageId,
812
1452
  model: model || options.model,
813
- timestamp,
814
- })
1453
+ timestamp: Date.now(),
1454
+ }
815
1455
  hasEmittedTextMessageStart = false
816
1456
  }
817
1457
 
@@ -819,43 +1459,62 @@ export class OpenAICompatibleResponsesTextAdapter<
819
1459
  // Otherwise surface incomplete_details.reason when present so
820
1460
  // callers can distinguish length-limit / content-filter cutoffs
821
1461
  // from a clean stop, mirroring the chat-completions adapter.
1462
+ // The Responses API's incomplete_details.reason ('max_output_tokens'
1463
+ // | 'content_filter') maps to the AG-UI finishReason vocabulary:
1464
+ // max_output_tokens → 'length', content_filter → 'content_filter'.
822
1465
  const hasFunctionCalls = chunk.response.output.some(
823
1466
  (item: unknown) =>
824
1467
  (item as { type: string }).type === 'function_call',
825
1468
  )
826
- const finishReason: string = hasFunctionCalls
1469
+ const incompleteReason = chunk.response.incomplete_details?.reason
1470
+ const finishReason:
1471
+ | 'tool_calls'
1472
+ | 'length'
1473
+ | 'content_filter'
1474
+ | 'stop' = hasFunctionCalls
827
1475
  ? 'tool_calls'
828
- : (chunk.response.incomplete_details?.reason ?? 'stop')
829
-
830
- yield asChunk({
831
- type: 'RUN_FINISHED',
1476
+ : incompleteReason === 'max_output_tokens'
1477
+ ? 'length'
1478
+ : incompleteReason === 'content_filter'
1479
+ ? 'content_filter'
1480
+ : 'stop'
1481
+
1482
+ yield {
1483
+ type: EventType.RUN_FINISHED,
832
1484
  runId: aguiState.runId,
1485
+ threadId: aguiState.threadId,
833
1486
  model: model || options.model,
834
- timestamp,
1487
+ timestamp: Date.now(),
835
1488
  usage: {
836
1489
  promptTokens: chunk.response.usage?.input_tokens || 0,
837
1490
  completionTokens: chunk.response.usage?.output_tokens || 0,
838
1491
  totalTokens: chunk.response.usage?.total_tokens || 0,
839
1492
  },
840
1493
  finishReason,
841
- })
1494
+ }
842
1495
  runFinishedEmitted = true
843
1496
  }
844
1497
 
845
1498
  if (chunk.type === 'error') {
846
- yield asChunk({
847
- type: 'RUN_ERROR',
848
- runId: aguiState.runId,
1499
+ yield {
1500
+ type: EventType.RUN_ERROR,
849
1501
  model: model || options.model,
850
- timestamp,
1502
+ timestamp: Date.now(),
1503
+ message: chunk.message,
1504
+ code: chunk.code ?? undefined,
851
1505
  error: {
852
1506
  message: chunk.message,
853
1507
  code: chunk.code ?? undefined,
854
1508
  },
855
- })
1509
+ }
856
1510
  // RUN_ERROR is terminal — don't let the synthetic RUN_FINISHED
857
- // block fire after a top-level stream error event.
1511
+ // block fire after a top-level stream error event, and stop
1512
+ // processing further chunks so no in-flight lifecycle events
1513
+ // (TEXT_MESSAGE_CONTENT, TOOL_CALL_*) leak past the terminal
1514
+ // error. Mirrors the `response.failed` / `response.incomplete`
1515
+ // branches above which return after their RUN_ERROR emission.
858
1516
  runFinishedEmitted = true
1517
+ return
859
1518
  }
860
1519
  }
861
1520
 
@@ -865,21 +1524,22 @@ export class OpenAICompatibleResponsesTextAdapter<
865
1524
  // see a terminal event for every started run.
866
1525
  if (!runFinishedEmitted && aguiState.hasEmittedRunStarted) {
867
1526
  if (hasEmittedTextMessageStart) {
868
- yield asChunk({
869
- type: 'TEXT_MESSAGE_END',
1527
+ yield {
1528
+ type: EventType.TEXT_MESSAGE_END,
870
1529
  messageId: aguiState.messageId,
871
1530
  model: model || options.model,
872
- timestamp,
873
- })
1531
+ timestamp: Date.now(),
1532
+ }
874
1533
  }
875
- yield asChunk({
876
- type: 'RUN_FINISHED',
1534
+ yield {
1535
+ type: EventType.RUN_FINISHED,
877
1536
  runId: aguiState.runId,
1537
+ threadId: aguiState.threadId,
878
1538
  model: model || options.model,
879
- timestamp,
1539
+ timestamp: Date.now(),
880
1540
  usage: undefined,
881
1541
  finishReason: toolCallMetadata.size > 0 ? 'tool_calls' : 'stop',
882
- })
1542
+ }
883
1543
  }
884
1544
  } catch (error: unknown) {
885
1545
  // Narrow before logging: raw SDK errors can carry request metadata
@@ -892,13 +1552,14 @@ export class OpenAICompatibleResponsesTextAdapter<
892
1552
  error: errorPayload,
893
1553
  source: `${this.name}.processStreamChunks`,
894
1554
  })
895
- yield asChunk({
896
- type: 'RUN_ERROR',
897
- runId: aguiState.runId,
1555
+ yield {
1556
+ type: EventType.RUN_ERROR,
898
1557
  model: options.model,
899
- timestamp,
1558
+ timestamp: Date.now(),
1559
+ message: errorPayload.message,
1560
+ code: errorPayload.code,
900
1561
  error: errorPayload,
901
- })
1562
+ }
902
1563
  }
903
1564
  }
904
1565
 
@@ -908,7 +1569,7 @@ export class OpenAICompatibleResponsesTextAdapter<
908
1569
  */
909
1570
  protected mapOptionsToRequest(
910
1571
  options: TextOptions<TProviderOptions>,
911
- ): Omit<OpenAI_SDK.Responses.ResponseCreateParams, 'stream'> {
1572
+ ): Omit<ResponseCreateParams, 'stream'> {
912
1573
  const input = this.convertMessagesToInput(options.messages)
913
1574
 
914
1575
  const tools = options.tools
@@ -961,8 +1622,8 @@ export class OpenAICompatibleResponsesTextAdapter<
961
1622
  */
962
1623
  protected convertMessagesToInput(
963
1624
  messages: Array<ModelMessage>,
964
- ): Responses.ResponseInput {
965
- const result: Responses.ResponseInput = []
1625
+ ): ResponseInput {
1626
+ const result: ResponseInput = []
966
1627
 
967
1628
  for (const message of messages) {
968
1629
  // Handle tool messages - convert to FunctionToolCallOutput
@@ -1016,7 +1677,7 @@ export class OpenAICompatibleResponsesTextAdapter<
1016
1677
 
1017
1678
  // Handle user messages (default case) — support multimodal content
1018
1679
  const contentParts = this.normalizeContent(message.content)
1019
- const inputContent: Array<Responses.ResponseInputContent> = []
1680
+ const inputContent: Array<ResponseInputContent> = []
1020
1681
 
1021
1682
  for (const part of contentParts) {
1022
1683
  inputContent.push(this.convertContentPartToInput(part))
@@ -1049,9 +1710,7 @@ export class OpenAICompatibleResponsesTextAdapter<
1049
1710
  * Handles text, image, and audio content parts.
1050
1711
  * Override this in subclasses for additional content types or provider-specific metadata.
1051
1712
  */
1052
- protected convertContentPartToInput(
1053
- part: ContentPart,
1054
- ): Responses.ResponseInputContent {
1713
+ protected convertContentPartToInput(part: ContentPart): ResponseInputContent {
1055
1714
  switch (part.type) {
1056
1715
  case 'text':
1057
1716
  return {