@tanstack/openai-base 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (110) hide show
  1. package/dist/esm/adapters/chat-completions-text.d.ts +76 -0
  2. package/dist/esm/adapters/chat-completions-text.js +411 -0
  3. package/dist/esm/adapters/chat-completions-text.js.map +1 -0
  4. package/dist/esm/adapters/chat-completions-tool-converter.d.ts +24 -0
  5. package/dist/esm/adapters/chat-completions-tool-converter.js +29 -0
  6. package/dist/esm/adapters/chat-completions-tool-converter.js.map +1 -0
  7. package/dist/esm/adapters/image.d.ts +32 -0
  8. package/dist/esm/adapters/image.js +69 -0
  9. package/dist/esm/adapters/image.js.map +1 -0
  10. package/dist/esm/adapters/responses-text.d.ts +115 -0
  11. package/dist/esm/adapters/responses-text.js +635 -0
  12. package/dist/esm/adapters/responses-text.js.map +1 -0
  13. package/dist/esm/adapters/responses-tool-converter.d.ts +35 -0
  14. package/dist/esm/adapters/responses-tool-converter.js +27 -0
  15. package/dist/esm/adapters/responses-tool-converter.js.map +1 -0
  16. package/dist/esm/adapters/summarize.d.ts +28 -0
  17. package/dist/esm/adapters/summarize.js +74 -0
  18. package/dist/esm/adapters/summarize.js.map +1 -0
  19. package/dist/esm/adapters/transcription.d.ts +39 -0
  20. package/dist/esm/adapters/transcription.js +139 -0
  21. package/dist/esm/adapters/transcription.js.map +1 -0
  22. package/dist/esm/adapters/tts.d.ts +26 -0
  23. package/dist/esm/adapters/tts.js +65 -0
  24. package/dist/esm/adapters/tts.js.map +1 -0
  25. package/dist/esm/adapters/video.d.ts +48 -0
  26. package/dist/esm/adapters/video.js +192 -0
  27. package/dist/esm/adapters/video.js.map +1 -0
  28. package/dist/esm/index.d.ts +15 -0
  29. package/dist/esm/index.js +65 -0
  30. package/dist/esm/index.js.map +1 -0
  31. package/dist/esm/tools/apply-patch-tool.d.ts +11 -0
  32. package/dist/esm/tools/apply-patch-tool.js +17 -0
  33. package/dist/esm/tools/apply-patch-tool.js.map +1 -0
  34. package/dist/esm/tools/code-interpreter-tool.d.ts +11 -0
  35. package/dist/esm/tools/code-interpreter-tool.js +22 -0
  36. package/dist/esm/tools/code-interpreter-tool.js.map +1 -0
  37. package/dist/esm/tools/computer-use-tool.d.ts +11 -0
  38. package/dist/esm/tools/computer-use-tool.js +23 -0
  39. package/dist/esm/tools/computer-use-tool.js.map +1 -0
  40. package/dist/esm/tools/custom-tool.d.ts +11 -0
  41. package/dist/esm/tools/custom-tool.js +23 -0
  42. package/dist/esm/tools/custom-tool.js.map +1 -0
  43. package/dist/esm/tools/file-search-tool.d.ts +11 -0
  44. package/dist/esm/tools/file-search-tool.js +30 -0
  45. package/dist/esm/tools/file-search-tool.js.map +1 -0
  46. package/dist/esm/tools/function-tool.d.ts +15 -0
  47. package/dist/esm/tools/function-tool.js +24 -0
  48. package/dist/esm/tools/function-tool.js.map +1 -0
  49. package/dist/esm/tools/image-generation-tool.d.ts +11 -0
  50. package/dist/esm/tools/image-generation-tool.js +27 -0
  51. package/dist/esm/tools/image-generation-tool.js.map +1 -0
  52. package/dist/esm/tools/index.d.ts +27 -0
  53. package/dist/esm/tools/local-shell-tool.d.ts +11 -0
  54. package/dist/esm/tools/local-shell-tool.js +17 -0
  55. package/dist/esm/tools/local-shell-tool.js.map +1 -0
  56. package/dist/esm/tools/mcp-tool.d.ts +12 -0
  57. package/dist/esm/tools/mcp-tool.js +31 -0
  58. package/dist/esm/tools/mcp-tool.js.map +1 -0
  59. package/dist/esm/tools/shell-tool.d.ts +11 -0
  60. package/dist/esm/tools/shell-tool.js +17 -0
  61. package/dist/esm/tools/shell-tool.js.map +1 -0
  62. package/dist/esm/tools/tool-choice.d.ts +17 -0
  63. package/dist/esm/tools/tool-converter.d.ts +6 -0
  64. package/dist/esm/tools/tool-converter.js +61 -0
  65. package/dist/esm/tools/tool-converter.js.map +1 -0
  66. package/dist/esm/tools/web-search-preview-tool.d.ts +11 -0
  67. package/dist/esm/tools/web-search-preview-tool.js +20 -0
  68. package/dist/esm/tools/web-search-preview-tool.js.map +1 -0
  69. package/dist/esm/tools/web-search-tool.d.ts +11 -0
  70. package/dist/esm/tools/web-search-tool.js +16 -0
  71. package/dist/esm/tools/web-search-tool.js.map +1 -0
  72. package/dist/esm/types/config.d.ts +4 -0
  73. package/dist/esm/types/message-metadata.d.ts +19 -0
  74. package/dist/esm/types/provider-options.d.ts +40 -0
  75. package/dist/esm/utils/client.d.ts +3 -0
  76. package/dist/esm/utils/client.js +8 -0
  77. package/dist/esm/utils/client.js.map +1 -0
  78. package/dist/esm/utils/schema-converter.d.ts +12 -0
  79. package/dist/esm/utils/schema-converter.js +65 -0
  80. package/dist/esm/utils/schema-converter.js.map +1 -0
  81. package/package.json +57 -0
  82. package/src/adapters/chat-completions-text.ts +817 -0
  83. package/src/adapters/chat-completions-tool-converter.ts +70 -0
  84. package/src/adapters/image.ts +158 -0
  85. package/src/adapters/responses-text.ts +1147 -0
  86. package/src/adapters/responses-tool-converter.ts +77 -0
  87. package/src/adapters/summarize.ts +174 -0
  88. package/src/adapters/transcription.ts +194 -0
  89. package/src/adapters/tts.ts +124 -0
  90. package/src/adapters/video.ts +385 -0
  91. package/src/index.ts +24 -0
  92. package/src/tools/apply-patch-tool.ts +32 -0
  93. package/src/tools/code-interpreter-tool.ts +39 -0
  94. package/src/tools/computer-use-tool.ts +38 -0
  95. package/src/tools/custom-tool.ts +33 -0
  96. package/src/tools/file-search-tool.ts +51 -0
  97. package/src/tools/function-tool.ts +44 -0
  98. package/src/tools/image-generation-tool.ts +51 -0
  99. package/src/tools/index.ts +41 -0
  100. package/src/tools/local-shell-tool.ts +32 -0
  101. package/src/tools/mcp-tool.ts +47 -0
  102. package/src/tools/shell-tool.ts +30 -0
  103. package/src/tools/tool-choice.ts +31 -0
  104. package/src/tools/tool-converter.ts +68 -0
  105. package/src/tools/web-search-preview-tool.ts +39 -0
  106. package/src/tools/web-search-tool.ts +38 -0
  107. package/src/types/config.ts +5 -0
  108. package/src/utils/client.ts +8 -0
  109. package/src/utils/request-options.ts +16 -0
  110. package/src/utils/schema-converter.ts +89 -0
@@ -0,0 +1,1147 @@
1
+ import { BaseTextAdapter } from '@tanstack/ai/adapters'
2
+ import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
3
+ import { generateId, transformNullsToUndefined } from '@tanstack/ai-utils'
4
+ import { createOpenAICompatibleClient } from '../utils/client'
5
+ import { extractRequestOptions } from '../utils/request-options'
6
+ import { makeStructuredOutputCompatible } from '../utils/schema-converter'
7
+ import { convertToolsToResponsesFormat } from './responses-tool-converter'
8
+ import type {
9
+ StructuredOutputOptions,
10
+ StructuredOutputResult,
11
+ } from '@tanstack/ai/adapters'
12
+ import type OpenAI_SDK from 'openai'
13
+ import type { Responses } from 'openai/resources'
14
+ import type {
15
+ ContentPart,
16
+ DefaultMessageMetadataByModality,
17
+ Modality,
18
+ ModelMessage,
19
+ StreamChunk,
20
+ TextOptions,
21
+ } from '@tanstack/ai'
22
+ import type { OpenAICompatibleClientConfig } from '../types/config'
23
+
24
+ /** Cast an event object to StreamChunk. Adapters construct events with string
25
+ * literal types which are structurally compatible with the EventType enum. */
26
+ const asChunk = (chunk: Record<string, unknown>) =>
27
+ chunk as unknown as StreamChunk
28
+
29
+ /**
30
+ * OpenAI-compatible Responses API Text Adapter
31
+ *
32
+ * A generalized base class for providers that use the OpenAI Responses API
33
+ * (`/v1/responses`). Providers like OpenAI (native), Azure OpenAI, and others
34
+ * that implement the Responses API can extend this class and only need to:
35
+ * - Set `baseURL` in the config
36
+ * - Lock the generic type parameters to provider-specific types
37
+ * - Override specific methods for quirks
38
+ *
39
+ * Key differences from the Chat Completions adapter:
40
+ * - Uses `client.responses.create()` instead of `client.chat.completions.create()`
41
+ * - Messages use `ResponseInput` format
42
+ * - System prompts go in `instructions` field, not as array messages
43
+ * - Streaming events are completely different (9+ event types vs simple delta chunks)
44
+ * - Supports reasoning/thinking tokens via `response.reasoning_text.delta`
45
+ * - Structured output uses `text.format` in the request (not `response_format`)
46
+ * - Tool calls use `response.function_call_arguments.delta`
47
+ * - Content parts are `input_text`, `input_image`, `input_file`
48
+ *
49
+ * All methods that build requests or process responses are `protected` so subclasses
50
+ * can override them.
51
+ */
52
+ export class OpenAICompatibleResponsesTextAdapter<
53
+ TModel extends string,
54
+ TProviderOptions extends Record<string, any> = Record<string, any>,
55
+ TInputModalities extends ReadonlyArray<Modality> = ReadonlyArray<Modality>,
56
+ TMessageMetadata extends DefaultMessageMetadataByModality =
57
+ DefaultMessageMetadataByModality,
58
+ TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,
59
+ > extends BaseTextAdapter<
60
+ TModel,
61
+ TProviderOptions,
62
+ TInputModalities,
63
+ TMessageMetadata,
64
+ TToolCapabilities
65
+ > {
66
+ readonly kind = 'text' as const
67
+ readonly name: string
68
+
69
+ protected client: OpenAI_SDK
70
+
71
+ constructor(
72
+ config: OpenAICompatibleClientConfig,
73
+ model: TModel,
74
+ name: string = 'openai-compatible-responses',
75
+ ) {
76
+ super({}, model)
77
+ this.name = name
78
+ this.client = createOpenAICompatibleClient(config)
79
+ }
80
+
81
+ async *chatStream(
82
+ options: TextOptions<TProviderOptions>,
83
+ ): AsyncIterable<StreamChunk> {
84
+ // Track tool call metadata by unique ID
85
+ // Responses API streams tool calls with deltas — first chunk has ID/name,
86
+ // subsequent chunks only have args.
87
+ // We assign our own indices as we encounter unique tool call IDs.
88
+ const toolCallMetadata = new Map<
89
+ string,
90
+ { index: number; name: string; started: boolean }
91
+ >()
92
+ const requestParams = this.mapOptionsToRequest(options)
93
+ const timestamp = Date.now()
94
+
95
+ // AG-UI lifecycle tracking
96
+ const aguiState = {
97
+ runId: generateId(this.name),
98
+ messageId: generateId(this.name),
99
+ timestamp,
100
+ hasEmittedRunStarted: false,
101
+ }
102
+
103
+ try {
104
+ options.logger.request(
105
+ `activity=chat provider=${this.name} model=${this.model} messages=${options.messages.length} tools=${options.tools?.length ?? 0} stream=true`,
106
+ { provider: this.name, model: this.model },
107
+ )
108
+ const response = await this.client.responses.create(
109
+ {
110
+ ...requestParams,
111
+ stream: true,
112
+ },
113
+ extractRequestOptions(options.request),
114
+ )
115
+
116
+ yield* this.processStreamChunks(
117
+ response,
118
+ toolCallMetadata,
119
+ options,
120
+ aguiState,
121
+ )
122
+ } catch (error: unknown) {
123
+ // Narrow before logging: raw SDK errors can carry request metadata
124
+ // (including auth headers) which we must never surface to user loggers.
125
+ const errorPayload = toRunErrorPayload(
126
+ error,
127
+ `${this.name}.chatStream failed`,
128
+ )
129
+
130
+ // Emit RUN_STARTED if not yet emitted
131
+ if (!aguiState.hasEmittedRunStarted) {
132
+ aguiState.hasEmittedRunStarted = true
133
+ yield asChunk({
134
+ type: 'RUN_STARTED',
135
+ runId: aguiState.runId,
136
+ model: options.model,
137
+ timestamp,
138
+ })
139
+ }
140
+
141
+ // Emit AG-UI RUN_ERROR
142
+ yield asChunk({
143
+ type: 'RUN_ERROR',
144
+ runId: aguiState.runId,
145
+ model: options.model,
146
+ timestamp,
147
+ error: errorPayload,
148
+ })
149
+
150
+ options.logger.errors(`${this.name}.chatStream fatal`, {
151
+ error: errorPayload,
152
+ source: `${this.name}.chatStream`,
153
+ })
154
+ }
155
+ }
156
+
157
+ /**
158
+ * Generate structured output using the provider's native JSON Schema response format.
159
+ * Uses stream: false to get the complete response in one call.
160
+ *
161
+ * OpenAI-compatible Responses APIs have strict requirements for structured output:
162
+ * - All properties must be in the `required` array
163
+ * - Optional fields should have null added to their type union
164
+ * - additionalProperties must be false for all objects
165
+ *
166
+ * The outputSchema is already JSON Schema (converted in the ai layer).
167
+ * We apply provider-specific transformations for structured output compatibility.
168
+ */
169
+ async structuredOutput(
170
+ options: StructuredOutputOptions<TProviderOptions>,
171
+ ): Promise<StructuredOutputResult<unknown>> {
172
+ const { chatOptions, outputSchema } = options
173
+ const requestParams = this.mapOptionsToRequest(chatOptions)
174
+
175
+ // Apply provider-specific transformations for structured output compatibility
176
+ const jsonSchema = this.makeStructuredOutputCompatible(
177
+ outputSchema,
178
+ outputSchema.required,
179
+ )
180
+
181
+ try {
182
+ // Strip streaming-only fields a subclass override of mapOptionsToRequest
183
+ // might have returned (parallel to chat-completions's structuredOutput
184
+ // cleanup) — sending stream_options to a non-streaming call is a 4xx.
185
+ const {
186
+ stream: _stream,
187
+ stream_options: _streamOptions,
188
+ ...cleanParams
189
+ } = requestParams as Record<string, unknown>
190
+ void _stream
191
+ void _streamOptions
192
+ chatOptions.logger.request(
193
+ `activity=structuredOutput provider=${this.name} model=${this.model} messages=${chatOptions.messages.length}`,
194
+ { provider: this.name, model: this.model },
195
+ )
196
+ const response = await this.client.responses.create(
197
+ {
198
+ ...(cleanParams as Omit<
199
+ OpenAI_SDK.Responses.ResponseCreateParams,
200
+ 'stream'
201
+ >),
202
+ stream: false,
203
+ // Configure structured output via text.format
204
+ text: {
205
+ format: {
206
+ type: 'json_schema',
207
+ name: 'structured_output',
208
+ schema: jsonSchema,
209
+ strict: true,
210
+ },
211
+ },
212
+ },
213
+ extractRequestOptions(chatOptions.request),
214
+ )
215
+
216
+ // Extract text content from the response. `stream: false` narrows the
217
+ // SDK return type to `Response`, but the explicit annotation makes
218
+ // that contract local rather than relying on inference through the
219
+ // overloaded `client.responses.create` signature.
220
+ const rawText = this.extractTextFromResponse(
221
+ response satisfies OpenAI_SDK.Responses.Response,
222
+ )
223
+
224
+ // Parse the JSON response
225
+ let parsed: unknown
226
+ try {
227
+ parsed = JSON.parse(rawText)
228
+ } catch {
229
+ throw new Error(
230
+ `Failed to parse structured output as JSON. Content: ${rawText.slice(0, 200)}${rawText.length > 200 ? '...' : ''}`,
231
+ )
232
+ }
233
+
234
+ // Transform null values to undefined to match original Zod schema expectations
235
+ // Provider returns null for optional fields we made nullable in the schema
236
+ const transformed = transformNullsToUndefined(parsed)
237
+
238
+ return {
239
+ data: transformed,
240
+ rawText,
241
+ }
242
+ } catch (error: unknown) {
243
+ // Narrow before logging: raw SDK errors can carry request metadata
244
+ // (including auth headers) which we must never surface to user loggers.
245
+ chatOptions.logger.errors(`${this.name}.structuredOutput fatal`, {
246
+ error: toRunErrorPayload(error, `${this.name}.structuredOutput failed`),
247
+ source: `${this.name}.structuredOutput`,
248
+ })
249
+ throw error
250
+ }
251
+ }
252
+
253
+ /**
254
+ * Applies provider-specific transformations for structured output compatibility.
255
+ * Override this in subclasses to handle provider-specific quirks.
256
+ */
257
+ protected makeStructuredOutputCompatible(
258
+ schema: Record<string, any>,
259
+ originalRequired?: Array<string>,
260
+ ): Record<string, any> {
261
+ return makeStructuredOutputCompatible(schema, originalRequired)
262
+ }
263
+
264
+ /**
265
+ * Extract text content from a non-streaming Responses API response.
266
+ * Override this in subclasses for provider-specific response shapes.
267
+ */
268
+ protected extractTextFromResponse(
269
+ response: OpenAI_SDK.Responses.Response,
270
+ ): string {
271
+ let textContent = ''
272
+ let refusal: string | undefined
273
+
274
+ for (const item of response.output) {
275
+ if (item.type === 'message') {
276
+ for (const part of item.content) {
277
+ if (part.type === 'output_text') {
278
+ textContent += part.text
279
+ } else {
280
+ // The Responses SDK currently models message content as
281
+ // `output_text | refusal`, so the only non-text branch is a
282
+ // refusal. Capture it so we can surface a distinct error below.
283
+ refusal = part.refusal || refusal || 'Refused without explanation'
284
+ }
285
+ }
286
+ }
287
+ }
288
+
289
+ // Surface refusals as an explicit error so callers don't see a generic
290
+ // "Failed to parse structured output as JSON. Content: " when the model
291
+ // refused for safety / content-policy reasons.
292
+ if (!textContent && refusal !== undefined) {
293
+ const err = new Error(`Model refused to respond: ${refusal}`)
294
+ ;(err as Error & { code?: string }).code = 'refusal'
295
+ throw err
296
+ }
297
+
298
+ return textContent
299
+ }
300
+
301
+ /**
302
+ * Processes streamed chunks from the Responses API and yields AG-UI events.
303
+ * Override this in subclasses to handle provider-specific stream behavior.
304
+ *
305
+ * Handles the following event types:
306
+ * - response.created / response.incomplete / response.failed
307
+ * - response.output_text.delta
308
+ * - response.reasoning_text.delta
309
+ * - response.reasoning_summary_text.delta
310
+ * - response.content_part.added / response.content_part.done
311
+ * - response.output_item.added
312
+ * - response.function_call_arguments.delta / response.function_call_arguments.done
313
+ * - response.completed
314
+ * - error
315
+ */
316
+ protected async *processStreamChunks(
317
+ stream: AsyncIterable<OpenAI_SDK.Responses.ResponseStreamEvent>,
318
+ toolCallMetadata: Map<
319
+ string,
320
+ { index: number; name: string; started: boolean }
321
+ >,
322
+ options: TextOptions<TProviderOptions>,
323
+ aguiState: {
324
+ runId: string
325
+ messageId: string
326
+ timestamp: number
327
+ hasEmittedRunStarted: boolean
328
+ },
329
+ ): AsyncIterable<StreamChunk> {
330
+ let accumulatedContent = ''
331
+ let accumulatedReasoning = ''
332
+ const timestamp = aguiState.timestamp
333
+
334
+ // Track if we've been streaming deltas to avoid duplicating content from done events
335
+ let hasStreamedContentDeltas = false
336
+ let hasStreamedReasoningDeltas = false
337
+
338
+ // Preserve response metadata across events
339
+ let model: string = options.model
340
+
341
+ // AG-UI lifecycle tracking
342
+ let stepId: string | null = null
343
+ let hasEmittedTextMessageStart = false
344
+ let hasEmittedStepStarted = false
345
+ // Track whether we've emitted a terminal RUN_FINISHED so the
346
+ // end-of-stream fallback below knows to synthesise one when the upstream
347
+ // cuts off without a response.completed event.
348
+ let runFinishedEmitted = false
349
+
350
+ try {
351
+ for await (const chunk of stream) {
352
+ options.logger.provider(`provider=${this.name} type=${chunk.type}`, {
353
+ provider: this.name,
354
+ type: chunk.type,
355
+ })
356
+
357
+ // Emit RUN_STARTED on first chunk
358
+ if (!aguiState.hasEmittedRunStarted) {
359
+ aguiState.hasEmittedRunStarted = true
360
+ yield asChunk({
361
+ type: 'RUN_STARTED',
362
+ runId: aguiState.runId,
363
+ model: model || options.model,
364
+ timestamp,
365
+ })
366
+ }
367
+
368
+ const handleContentPart = (contentPart: {
369
+ type: string
370
+ text?: string
371
+ refusal?: string
372
+ }): StreamChunk => {
373
+ if (contentPart.type === 'output_text') {
374
+ accumulatedContent += contentPart.text || ''
375
+ return asChunk({
376
+ type: 'TEXT_MESSAGE_CONTENT',
377
+ messageId: aguiState.messageId,
378
+ model: model || options.model,
379
+ timestamp,
380
+ delta: contentPart.text || '',
381
+ content: accumulatedContent,
382
+ })
383
+ }
384
+
385
+ if (contentPart.type === 'reasoning_text') {
386
+ accumulatedReasoning += contentPart.text || ''
387
+ // Cache the fallback stepId rather than generating a fresh one
388
+ // on every call — otherwise multiple reasoning chunks arriving
389
+ // before STEP_STARTED was emitted (e.g. via response.content_part.done
390
+ // alone) would each get a different stepId and break correlation.
391
+ if (!stepId) {
392
+ stepId = generateId(this.name)
393
+ }
394
+ return asChunk({
395
+ type: 'STEP_FINISHED',
396
+ stepId,
397
+ model: model || options.model,
398
+ timestamp,
399
+ delta: contentPart.text || '',
400
+ content: accumulatedReasoning,
401
+ })
402
+ }
403
+ // Either a real refusal or an unknown content_part type. Surface
404
+ // the part type in the error so unknown parts are debuggable
405
+ // instead of being misreported as "Unknown refusal".
406
+ const isRefusal = contentPart.type === 'refusal'
407
+ const message = isRefusal
408
+ ? contentPart.refusal || 'Refused without explanation'
409
+ : `Unsupported response content_part type: ${contentPart.type}`
410
+ return asChunk({
411
+ type: 'RUN_ERROR',
412
+ runId: aguiState.runId,
413
+ model: model || options.model,
414
+ timestamp,
415
+ error: {
416
+ message,
417
+ code: isRefusal ? 'refusal' : contentPart.type,
418
+ },
419
+ })
420
+ }
421
+
422
+ // Capture model metadata from any of these events (created starts
423
+ // the run; failed/incomplete signal terminal failure).
424
+ if (
425
+ chunk.type === 'response.created' ||
426
+ chunk.type === 'response.incomplete' ||
427
+ chunk.type === 'response.failed'
428
+ ) {
429
+ model = chunk.response.model
430
+ }
431
+
432
+ // response.created marks the start of a fresh run — safe to reset
433
+ // the per-run accumulators here.
434
+ if (chunk.type === 'response.created') {
435
+ hasStreamedContentDeltas = false
436
+ hasStreamedReasoningDeltas = false
437
+ hasEmittedTextMessageStart = false
438
+ hasEmittedStepStarted = false
439
+ accumulatedContent = ''
440
+ accumulatedReasoning = ''
441
+ }
442
+
443
+ // response.failed and response.incomplete are TERMINAL events for
444
+ // the current response. Close any open AG-UI message lifecycle FIRST
445
+ // so consumers tracking start/end pairs don't see an unbalanced
446
+ // TEXT_MESSAGE_START. Then surface the error and mark the run as
447
+ // finished so the post-loop synthetic terminal block doesn't emit
448
+ // a duplicate RUN_FINISHED on top of RUN_ERROR.
449
+ if (
450
+ chunk.type === 'response.failed' ||
451
+ chunk.type === 'response.incomplete'
452
+ ) {
453
+ if (hasEmittedTextMessageStart) {
454
+ yield asChunk({
455
+ type: 'TEXT_MESSAGE_END',
456
+ messageId: aguiState.messageId,
457
+ model: chunk.response.model,
458
+ timestamp,
459
+ })
460
+ hasEmittedTextMessageStart = false
461
+ }
462
+ // Coalesce error + incomplete_details into a single RUN_ERROR
463
+ // payload — emitting two distinct events for one terminal upstream
464
+ // event would force consumers to handle a non-existent ordering.
465
+ const errorMessage =
466
+ chunk.response.error?.message ||
467
+ chunk.response.incomplete_details?.reason ||
468
+ (chunk.type === 'response.failed'
469
+ ? 'Response failed'
470
+ : 'Response ended incomplete')
471
+ const errorCode =
472
+ chunk.response.error?.code ||
473
+ (chunk.response.incomplete_details ? 'incomplete' : undefined)
474
+ // Always emit RUN_ERROR for terminal failure events, even when the
475
+ // upstream omitted both `error` and `incomplete_details`. Skipping
476
+ // emission on a `response.incomplete` with no detail would let the
477
+ // post-loop synthetic block silently coerce the run to a clean
478
+ // `RUN_FINISHED { finishReason: 'stop' }` — masking the failure.
479
+ yield asChunk({
480
+ type: 'RUN_ERROR',
481
+ runId: aguiState.runId,
482
+ model: chunk.response.model,
483
+ timestamp,
484
+ error: {
485
+ message: errorMessage,
486
+ ...(errorCode !== undefined && { code: errorCode }),
487
+ },
488
+ })
489
+ // RUN_ERROR is the terminal event for this run; stop processing
490
+ // any further chunks the iterator might still deliver.
491
+ runFinishedEmitted = true
492
+ return
493
+ }
494
+
495
+ // Handle output text deltas (token-by-token streaming)
496
+ // response.output_text.delta provides incremental text updates
497
+ if (chunk.type === 'response.output_text.delta' && chunk.delta) {
498
+ // Delta can be an array of strings or a single string
499
+ const textDelta = Array.isArray(chunk.delta)
500
+ ? chunk.delta.join('')
501
+ : typeof chunk.delta === 'string'
502
+ ? chunk.delta
503
+ : ''
504
+
505
+ if (textDelta) {
506
+ // Emit TEXT_MESSAGE_START on first text content
507
+ if (!hasEmittedTextMessageStart) {
508
+ hasEmittedTextMessageStart = true
509
+ yield asChunk({
510
+ type: 'TEXT_MESSAGE_START',
511
+ messageId: aguiState.messageId,
512
+ model: model || options.model,
513
+ timestamp,
514
+ role: 'assistant',
515
+ })
516
+ }
517
+
518
+ accumulatedContent += textDelta
519
+ hasStreamedContentDeltas = true
520
+ yield asChunk({
521
+ type: 'TEXT_MESSAGE_CONTENT',
522
+ messageId: aguiState.messageId,
523
+ model: model || options.model,
524
+ timestamp,
525
+ delta: textDelta,
526
+ content: accumulatedContent,
527
+ })
528
+ }
529
+ }
530
+
531
+ // Handle reasoning deltas (token-by-token thinking/reasoning streaming)
532
+ // response.reasoning_text.delta provides incremental reasoning updates
533
+ if (chunk.type === 'response.reasoning_text.delta' && chunk.delta) {
534
+ // Delta can be an array of strings or a single string
535
+ const reasoningDelta = Array.isArray(chunk.delta)
536
+ ? chunk.delta.join('')
537
+ : typeof chunk.delta === 'string'
538
+ ? chunk.delta
539
+ : ''
540
+
541
+ if (reasoningDelta) {
542
+ // Emit STEP_STARTED on first reasoning content
543
+ if (!hasEmittedStepStarted) {
544
+ hasEmittedStepStarted = true
545
+ stepId = generateId(this.name)
546
+ yield asChunk({
547
+ type: 'STEP_STARTED',
548
+ stepId,
549
+ model: model || options.model,
550
+ timestamp,
551
+ stepType: 'thinking',
552
+ })
553
+ }
554
+
555
+ accumulatedReasoning += reasoningDelta
556
+ hasStreamedReasoningDeltas = true
557
+ yield asChunk({
558
+ type: 'STEP_FINISHED',
559
+ stepId: stepId || generateId(this.name),
560
+ model: model || options.model,
561
+ timestamp,
562
+ delta: reasoningDelta,
563
+ content: accumulatedReasoning,
564
+ })
565
+ }
566
+ }
567
+
568
+ // Handle reasoning summary deltas (when using reasoning.summary option)
569
+ // response.reasoning_summary_text.delta provides incremental summary updates
570
+ if (
571
+ chunk.type === 'response.reasoning_summary_text.delta' &&
572
+ chunk.delta
573
+ ) {
574
+ const summaryDelta =
575
+ typeof chunk.delta === 'string' ? chunk.delta : ''
576
+
577
+ if (summaryDelta) {
578
+ // Emit STEP_STARTED on first reasoning content
579
+ if (!hasEmittedStepStarted) {
580
+ hasEmittedStepStarted = true
581
+ stepId = generateId(this.name)
582
+ yield asChunk({
583
+ type: 'STEP_STARTED',
584
+ stepId,
585
+ model: model || options.model,
586
+ timestamp,
587
+ stepType: 'thinking',
588
+ })
589
+ }
590
+
591
+ accumulatedReasoning += summaryDelta
592
+ hasStreamedReasoningDeltas = true
593
+ yield asChunk({
594
+ type: 'STEP_FINISHED',
595
+ stepId: stepId || generateId(this.name),
596
+ model: model || options.model,
597
+ timestamp,
598
+ delta: summaryDelta,
599
+ content: accumulatedReasoning,
600
+ })
601
+ }
602
+ }
603
+
604
+ // handle content_part added events for text, reasoning and refusals
605
+ if (chunk.type === 'response.content_part.added') {
606
+ const contentPart = chunk.part
607
+ // Emit TEXT_MESSAGE_START if this is text content
608
+ if (
609
+ contentPart.type === 'output_text' &&
610
+ !hasEmittedTextMessageStart
611
+ ) {
612
+ hasEmittedTextMessageStart = true
613
+ yield asChunk({
614
+ type: 'TEXT_MESSAGE_START',
615
+ messageId: aguiState.messageId,
616
+ model: model || options.model,
617
+ timestamp,
618
+ role: 'assistant',
619
+ })
620
+ }
621
+ // Emit STEP_STARTED if this is reasoning content
622
+ if (contentPart.type === 'reasoning_text' && !hasEmittedStepStarted) {
623
+ hasEmittedStepStarted = true
624
+ stepId = generateId(this.name)
625
+ yield asChunk({
626
+ type: 'STEP_STARTED',
627
+ stepId,
628
+ model: model || options.model,
629
+ timestamp,
630
+ stepType: 'thinking',
631
+ })
632
+ }
633
+ // Mark whichever stream we just emitted into so a subsequent
634
+ // `content_part.done` doesn't duplicate the same text. Without
635
+ // this flag, an `added` event carrying the full text followed by
636
+ // a matching `done` event would emit TEXT_MESSAGE_CONTENT twice.
637
+ if (contentPart.type === 'output_text') {
638
+ hasStreamedContentDeltas = true
639
+ } else if (contentPart.type === 'reasoning_text') {
640
+ hasStreamedReasoningDeltas = true
641
+ }
642
+ const partChunk = handleContentPart(contentPart)
643
+ yield partChunk
644
+ // handleContentPart returns RUN_ERROR for refusals / unknown
645
+ // content_part types — those are terminal events. Don't keep
646
+ // processing more chunks (and don't let the post-loop synthetic
647
+ // block emit a second terminal event).
648
+ if (partChunk.type === 'RUN_ERROR') {
649
+ runFinishedEmitted = true
650
+ return
651
+ }
652
+ }
653
+
654
+ if (chunk.type === 'response.content_part.done') {
655
+ const contentPart = chunk.part
656
+
657
+ // Skip emitting chunks for content parts that we've already streamed via deltas
658
+ // The done event is just a completion marker, not new content
659
+ if (contentPart.type === 'output_text' && hasStreamedContentDeltas) {
660
+ // Content already accumulated from deltas, skip
661
+ continue
662
+ }
663
+ if (
664
+ contentPart.type === 'reasoning_text' &&
665
+ hasStreamedReasoningDeltas
666
+ ) {
667
+ // Reasoning already accumulated from deltas, skip
668
+ continue
669
+ }
670
+
671
+ // Only emit if we haven't been streaming deltas (e.g., for non-streaming responses)
672
+ const doneChunk = handleContentPart(contentPart)
673
+ yield doneChunk
674
+ if (doneChunk.type === 'RUN_ERROR') {
675
+ runFinishedEmitted = true
676
+ return
677
+ }
678
+ }
679
+
680
+ // handle output_item.added to capture function call metadata (name)
681
+ if (chunk.type === 'response.output_item.added') {
682
+ const item = chunk.item
683
+ if (item.type === 'function_call' && item.id) {
684
+ const existing = toolCallMetadata.get(item.id)
685
+ // Only emit TOOL_CALL_START on the FIRST output_item.added for
686
+ // an item id. A duplicate emission (which can happen on retried
687
+ // streams or replay) would violate AG-UI's start-once contract.
688
+ if (!existing?.started) {
689
+ if (!existing) {
690
+ toolCallMetadata.set(item.id, {
691
+ index: chunk.output_index,
692
+ name: item.name || '',
693
+ started: false,
694
+ })
695
+ }
696
+ yield asChunk({
697
+ type: 'TOOL_CALL_START',
698
+ toolCallId: item.id,
699
+ toolCallName: item.name || '',
700
+ toolName: item.name || '',
701
+ model: model || options.model,
702
+ timestamp,
703
+ index: chunk.output_index,
704
+ })
705
+ toolCallMetadata.get(item.id)!.started = true
706
+ }
707
+ }
708
+ }
709
+
710
+ // Handle function call arguments delta (streaming). Drop the
711
+ // previously-emitted `args` field — it had inverted polarity
712
+ // (populated only when metadata was MISSING, i.e. when the
713
+ // matching TOOL_CALL_START hadn't fired) and the chat-completions
714
+ // adapter never emitted it, so it leaked partial deltas as
715
+ // pseudo-args only on the orphan path. Consumers should accumulate
716
+ // `delta` themselves.
717
+ //
718
+ // Guard with `metadata?.started`: the matching TOOL_CALL_START fires
719
+ // from `output_item.added`, and emitting TOOL_CALL_ARGS before that
720
+ // would violate the AG-UI lifecycle (ARGS without START). The .done
721
+ // handler below applies the same guard.
722
+ if (
723
+ chunk.type === 'response.function_call_arguments.delta' &&
724
+ chunk.delta
725
+ ) {
726
+ const metadata = toolCallMetadata.get(chunk.item_id)
727
+ if (!metadata?.started) {
728
+ options.logger.errors(
729
+ `${this.name}.processStreamChunks orphan function_call_arguments.delta`,
730
+ {
731
+ source: `${this.name}.processStreamChunks`,
732
+ toolCallId: chunk.item_id,
733
+ rawDelta: chunk.delta,
734
+ },
735
+ )
736
+ continue
737
+ }
738
+ yield asChunk({
739
+ type: 'TOOL_CALL_ARGS',
740
+ toolCallId: chunk.item_id,
741
+ model: model || options.model,
742
+ timestamp,
743
+ delta: chunk.delta,
744
+ })
745
+ }
746
+
747
+ if (chunk.type === 'response.function_call_arguments.done') {
748
+ const { item_id } = chunk
749
+
750
+ // Get the function name from metadata (captured in output_item.added)
751
+ const metadata = toolCallMetadata.get(item_id)
752
+ // Skip TOOL_CALL_END for items whose start was never emitted (no
753
+ // matching `output_item.added`). Emitting END without START would
754
+ // produce an unbalanced AG-UI lifecycle event downstream consumers
755
+ // can't pair.
756
+ if (!metadata?.started) {
757
+ options.logger.errors(
758
+ `${this.name}.processStreamChunks orphan function_call_arguments.done`,
759
+ {
760
+ source: `${this.name}.processStreamChunks`,
761
+ toolCallId: item_id,
762
+ rawArguments: chunk.arguments,
763
+ },
764
+ )
765
+ continue
766
+ }
767
+ const name = metadata.name || ''
768
+
769
+ // Parse arguments. Surface parse failures via the logger so a
770
+ // model emitting malformed JSON is debuggable instead of silently
771
+ // invoking the tool with {}.
772
+ let parsedInput: unknown = {}
773
+ if (chunk.arguments) {
774
+ try {
775
+ const parsed = JSON.parse(chunk.arguments)
776
+ parsedInput = parsed && typeof parsed === 'object' ? parsed : {}
777
+ } catch (parseError) {
778
+ options.logger.errors(
779
+ `${this.name}.processStreamChunks tool-args JSON parse failed`,
780
+ {
781
+ error: toRunErrorPayload(
782
+ parseError,
783
+ `tool ${name} (${item_id}) returned malformed JSON arguments`,
784
+ ),
785
+ source: `${this.name}.processStreamChunks`,
786
+ toolCallId: item_id,
787
+ toolName: name,
788
+ rawArguments: chunk.arguments,
789
+ },
790
+ )
791
+ parsedInput = {}
792
+ }
793
+ }
794
+
795
+ yield asChunk({
796
+ type: 'TOOL_CALL_END',
797
+ toolCallId: item_id,
798
+ toolCallName: name,
799
+ toolName: name,
800
+ model: model || options.model,
801
+ timestamp,
802
+ input: parsedInput,
803
+ })
804
+ }
805
+
806
+ if (chunk.type === 'response.completed') {
807
+ // Emit TEXT_MESSAGE_END if we had text content
808
+ if (hasEmittedTextMessageStart) {
809
+ yield asChunk({
810
+ type: 'TEXT_MESSAGE_END',
811
+ messageId: aguiState.messageId,
812
+ model: model || options.model,
813
+ timestamp,
814
+ })
815
+ hasEmittedTextMessageStart = false
816
+ }
817
+
818
+ // Determine finish reason. Function-call output → tool_calls.
819
+ // Otherwise surface incomplete_details.reason when present so
820
+ // callers can distinguish length-limit / content-filter cutoffs
821
+ // from a clean stop, mirroring the chat-completions adapter.
822
+ const hasFunctionCalls = chunk.response.output.some(
823
+ (item: unknown) =>
824
+ (item as { type: string }).type === 'function_call',
825
+ )
826
+ const finishReason: string = hasFunctionCalls
827
+ ? 'tool_calls'
828
+ : (chunk.response.incomplete_details?.reason ?? 'stop')
829
+
830
+ yield asChunk({
831
+ type: 'RUN_FINISHED',
832
+ runId: aguiState.runId,
833
+ model: model || options.model,
834
+ timestamp,
835
+ usage: {
836
+ promptTokens: chunk.response.usage?.input_tokens || 0,
837
+ completionTokens: chunk.response.usage?.output_tokens || 0,
838
+ totalTokens: chunk.response.usage?.total_tokens || 0,
839
+ },
840
+ finishReason,
841
+ })
842
+ runFinishedEmitted = true
843
+ }
844
+
845
+ if (chunk.type === 'error') {
846
+ yield asChunk({
847
+ type: 'RUN_ERROR',
848
+ runId: aguiState.runId,
849
+ model: model || options.model,
850
+ timestamp,
851
+ error: {
852
+ message: chunk.message,
853
+ code: chunk.code ?? undefined,
854
+ },
855
+ })
856
+ // RUN_ERROR is terminal — don't let the synthetic RUN_FINISHED
857
+ // block fire after a top-level stream error event.
858
+ runFinishedEmitted = true
859
+ }
860
+ }
861
+
862
+ // Synthetic terminal RUN_FINISHED if the stream ended without a
863
+ // response.completed event (e.g. truncated upstream connection). This
864
+ // mirrors the chat-completions adapter's behavior so consumers always
865
+ // see a terminal event for every started run.
866
+ if (!runFinishedEmitted && aguiState.hasEmittedRunStarted) {
867
+ if (hasEmittedTextMessageStart) {
868
+ yield asChunk({
869
+ type: 'TEXT_MESSAGE_END',
870
+ messageId: aguiState.messageId,
871
+ model: model || options.model,
872
+ timestamp,
873
+ })
874
+ }
875
+ yield asChunk({
876
+ type: 'RUN_FINISHED',
877
+ runId: aguiState.runId,
878
+ model: model || options.model,
879
+ timestamp,
880
+ usage: undefined,
881
+ finishReason: toolCallMetadata.size > 0 ? 'tool_calls' : 'stop',
882
+ })
883
+ }
884
+ } catch (error: unknown) {
885
+ // Narrow before logging: raw SDK errors can carry request metadata
886
+ // (including auth headers) which we must never surface to user loggers.
887
+ const errorPayload = toRunErrorPayload(
888
+ error,
889
+ `${this.name}.processStreamChunks failed`,
890
+ )
891
+ options.logger.errors(`${this.name}.processStreamChunks fatal`, {
892
+ error: errorPayload,
893
+ source: `${this.name}.processStreamChunks`,
894
+ })
895
+ yield asChunk({
896
+ type: 'RUN_ERROR',
897
+ runId: aguiState.runId,
898
+ model: options.model,
899
+ timestamp,
900
+ error: errorPayload,
901
+ })
902
+ }
903
+ }
904
+
905
+ /**
906
+ * Maps common TextOptions to Responses API request format.
907
+ * Override this in subclasses to add provider-specific options.
908
+ */
909
+ protected mapOptionsToRequest(
910
+ options: TextOptions<TProviderOptions>,
911
+ ): Omit<OpenAI_SDK.Responses.ResponseCreateParams, 'stream'> {
912
+ const input = this.convertMessagesToInput(options.messages)
913
+
914
+ const tools = options.tools
915
+ ? convertToolsToResponsesFormat(
916
+ options.tools,
917
+ this.makeStructuredOutputCompatible.bind(this),
918
+ )
919
+ : undefined
920
+
921
+ const modelOptions = options.modelOptions
922
+
923
+ // Spread modelOptions first, then explicit top-level options when set.
924
+ // Mirrors the chat-completions base adapter's precedence so callers
925
+ // tuning either backend get identical behaviour. Leaving `modelOptions`
926
+ // last (its previous behavior) silently shadowed the canonical
927
+ // `options.temperature`/`maxTokens` fields, while spreading first
928
+ // without nullish-aware merge would clobber `modelOptions.temperature`
929
+ // with `undefined` whenever the caller didn't set the top-level option.
930
+ return {
931
+ ...modelOptions,
932
+ model: options.model,
933
+ ...(options.temperature !== undefined && {
934
+ temperature: options.temperature,
935
+ }),
936
+ ...(options.maxTokens !== undefined && {
937
+ max_output_tokens: options.maxTokens,
938
+ }),
939
+ ...(options.topP !== undefined && { top_p: options.topP }),
940
+ ...(options.metadata !== undefined && { metadata: options.metadata }),
941
+ ...(options.systemPrompts &&
942
+ options.systemPrompts.length > 0 && {
943
+ instructions: options.systemPrompts.join('\n'),
944
+ }),
945
+ input,
946
+ // Conditional spread: `tools: undefined` would clobber any
947
+ // modelOptions.tools the caller set above.
948
+ ...(tools && tools.length > 0 && { tools }),
949
+ }
950
+ }
951
+
952
+ /**
953
+ * Converts ModelMessage[] to Responses API ResponseInput format.
954
+ * Override this in subclasses for provider-specific message format quirks.
955
+ *
956
+ * Key differences from Chat Completions:
957
+ * - Tool results use `function_call_output` type (not `tool` role)
958
+ * - Assistant tool calls are `function_call` objects (not nested in `tool_calls`)
959
+ * - User content uses `input_text`, `input_image`, `input_file` types
960
+ * - System prompts go in `instructions`, not as messages
961
+ */
962
+ protected convertMessagesToInput(
963
+ messages: Array<ModelMessage>,
964
+ ): Responses.ResponseInput {
965
+ const result: Responses.ResponseInput = []
966
+
967
+ for (const message of messages) {
968
+ // Handle tool messages - convert to FunctionToolCallOutput
969
+ if (message.role === 'tool') {
970
+ result.push({
971
+ type: 'function_call_output',
972
+ call_id: message.toolCallId || '',
973
+ output:
974
+ typeof message.content === 'string'
975
+ ? message.content
976
+ : JSON.stringify(message.content),
977
+ })
978
+ continue
979
+ }
980
+
981
+ // Handle assistant messages
982
+ if (message.role === 'assistant') {
983
+ // If the assistant message has tool calls, add them as FunctionToolCall objects
984
+ // Responses API expects arguments as a string (JSON string)
985
+ if (message.toolCalls && message.toolCalls.length > 0) {
986
+ for (const toolCall of message.toolCalls) {
987
+ // Keep arguments as string for Responses API
988
+ const argumentsString =
989
+ typeof toolCall.function.arguments === 'string'
990
+ ? toolCall.function.arguments
991
+ : JSON.stringify(toolCall.function.arguments)
992
+
993
+ result.push({
994
+ type: 'function_call',
995
+ call_id: toolCall.id,
996
+ name: toolCall.function.name,
997
+ arguments: argumentsString,
998
+ })
999
+ }
1000
+ }
1001
+
1002
+ // Add the assistant's text message if there is content
1003
+ if (message.content) {
1004
+ const contentStr = this.extractTextContent(message.content)
1005
+ if (contentStr) {
1006
+ result.push({
1007
+ type: 'message',
1008
+ role: 'assistant',
1009
+ content: contentStr,
1010
+ })
1011
+ }
1012
+ }
1013
+
1014
+ continue
1015
+ }
1016
+
1017
+ // Handle user messages (default case) — support multimodal content
1018
+ const contentParts = this.normalizeContent(message.content)
1019
+ const inputContent: Array<Responses.ResponseInputContent> = []
1020
+
1021
+ for (const part of contentParts) {
1022
+ inputContent.push(this.convertContentPartToInput(part))
1023
+ }
1024
+
1025
+ if (inputContent.length === 0) {
1026
+ // Fail loud rather than silently sending an empty user message —
1027
+ // mirrors the chat-completions adapter, where a paid-but-empty
1028
+ // request would mask the real intent (caller passed `null` content
1029
+ // or a normalize step dropped everything).
1030
+ throw new Error(
1031
+ `User message for ${this.name} has no content parts. ` +
1032
+ `Empty user messages would produce a paid request with no input; ` +
1033
+ `provide at least one text/image/audio part or omit the message.`,
1034
+ )
1035
+ }
1036
+
1037
+ result.push({
1038
+ type: 'message',
1039
+ role: 'user',
1040
+ content: inputContent,
1041
+ })
1042
+ }
1043
+
1044
+ return result
1045
+ }
1046
+
1047
+ /**
1048
+ * Converts a ContentPart to Responses API input content item.
1049
+ * Handles text, image, and audio content parts.
1050
+ * Override this in subclasses for additional content types or provider-specific metadata.
1051
+ */
1052
+ protected convertContentPartToInput(
1053
+ part: ContentPart,
1054
+ ): Responses.ResponseInputContent {
1055
+ switch (part.type) {
1056
+ case 'text':
1057
+ return {
1058
+ type: 'input_text',
1059
+ text: part.content,
1060
+ }
1061
+ case 'image': {
1062
+ const imageMetadata = part.metadata as
1063
+ | { detail?: 'auto' | 'low' | 'high' }
1064
+ | undefined
1065
+ if (part.source.type === 'url') {
1066
+ return {
1067
+ type: 'input_image',
1068
+ image_url: part.source.value,
1069
+ detail: imageMetadata?.detail || 'auto',
1070
+ }
1071
+ }
1072
+ // For base64 data, construct a data URI using the mimeType from
1073
+ // source. Default to a generic octet-stream MIME if the source
1074
+ // didn't supply one — letting `undefined` interpolate would produce
1075
+ // an invalid URI like "data:undefined;base64,...".
1076
+ const imageValue = part.source.value
1077
+ const imageMime = part.source.mimeType || 'application/octet-stream'
1078
+ const imageUrl = imageValue.startsWith('data:')
1079
+ ? imageValue
1080
+ : `data:${imageMime};base64,${imageValue}`
1081
+ return {
1082
+ type: 'input_image',
1083
+ image_url: imageUrl,
1084
+ detail: imageMetadata?.detail || 'auto',
1085
+ }
1086
+ }
1087
+ case 'audio': {
1088
+ if (part.source.type === 'url') {
1089
+ return {
1090
+ type: 'input_file',
1091
+ file_url: part.source.value,
1092
+ }
1093
+ }
1094
+ // Wrap raw base64 in a data URL — `input_file` rejects bare base64
1095
+ // payloads (matches the image branch above which already does this).
1096
+ // Default the MIME if missing so we never interpolate `undefined`.
1097
+ const audioValue = part.source.value
1098
+ const audioMime = part.source.mimeType || 'application/octet-stream'
1099
+ const audioFileData = audioValue.startsWith('data:')
1100
+ ? audioValue
1101
+ : `data:${audioMime};base64,${audioValue}`
1102
+ return {
1103
+ type: 'input_file',
1104
+ file_data: audioFileData,
1105
+ }
1106
+ }
1107
+
1108
+ default:
1109
+ throw new Error(`Unsupported content part type: ${part.type}`)
1110
+ }
1111
+ }
1112
+
1113
+ /**
1114
+ * Normalizes message content to an array of ContentPart.
1115
+ * Handles backward compatibility with string content.
1116
+ */
1117
+ protected normalizeContent(
1118
+ content: string | null | Array<ContentPart>,
1119
+ ): Array<ContentPart> {
1120
+ if (content === null) {
1121
+ return []
1122
+ }
1123
+ if (typeof content === 'string') {
1124
+ return [{ type: 'text', content: content }]
1125
+ }
1126
+ return content
1127
+ }
1128
+
1129
+ /**
1130
+ * Extracts text content from a content value that may be string, null, or ContentPart array.
1131
+ */
1132
+ protected extractTextContent(
1133
+ content: string | null | Array<ContentPart>,
1134
+ ): string {
1135
+ if (content === null) {
1136
+ return ''
1137
+ }
1138
+ if (typeof content === 'string') {
1139
+ return content
1140
+ }
1141
+ // It's an array of ContentPart
1142
+ return content
1143
+ .filter((p) => p.type === 'text')
1144
+ .map((p) => p.content)
1145
+ .join('')
1146
+ }
1147
+ }