@tanstack/openai-base 0.10.16 → 0.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,4 +1,10 @@
1
- import { EventType, normalizeSystemPrompts } from '@tanstack/ai'
1
+ import {
2
+ EventType,
3
+ fileReferenceFor,
4
+ isFileSource,
5
+ normalizeSystemPrompts,
6
+ unsupportedFileSourceError,
7
+ } from '@tanstack/ai'
2
8
  import { BaseTextAdapter } from '@tanstack/ai/adapters'
3
9
  import {
4
10
  toRunErrorPayload,
@@ -6,11 +12,24 @@ import {
6
12
  } from '@tanstack/ai/adapter-internals'
7
13
  import { generateId } from '@tanstack/ai-utils'
8
14
  import { extractRequestOptions } from '../utils/request-options'
9
- import { makeStructuredOutputCompatibleWithMap } from '../utils/schema-converter'
15
+ import {
16
+ makeStructuredOutputCompatibleWithMap,
17
+ warnStrictFallback,
18
+ } from '../utils/schema-converter'
10
19
  import { createToolInputNormalizer } from '../utils/tool-input-normalizer'
11
- import type { StructuredOutputCompatibility } from '../utils/schema-converter'
20
+ import type {
21
+ OpenAIBaseTextAdapterOptions,
22
+ StructuredOutputCompatibility,
23
+ } from '../utils/schema-converter'
12
24
  import { buildResponsesUsage } from '../usage'
13
25
  import { convertToolsToResponsesFormat } from './responses-tool-converter'
26
+ import {
27
+ hostedShellCallIds,
28
+ readUserExecutedCall,
29
+ userToolRequestItem,
30
+ userToolResultItem,
31
+ } from './responses-user-tools'
32
+ import type { OpenAIUserToolName } from './responses-user-tools'
14
33
  import type OpenAI from 'openai'
15
34
  import type {
16
35
  StructuredOutputOptions,
@@ -20,8 +39,11 @@ import type {
20
39
  Response,
21
40
  ResponseCreateParams,
22
41
  ResponseFunctionCallOutputItem,
42
+ ResponseFunctionWebSearch,
23
43
  ResponseInput,
24
44
  ResponseInputContent,
45
+ ResponseOutputMessage,
46
+ ResponseOutputText,
25
47
  ResponseStreamEvent,
26
48
  } from 'openai/resources/responses/responses'
27
49
  import type {
@@ -30,6 +52,8 @@ import type {
30
52
  Modality,
31
53
  ModelMessage,
32
54
  AdapterYieldChunk,
55
+ ProviderExecutedToolMetadata,
56
+ ProviderExecutedToolSource,
33
57
  TextOptions,
34
58
  } from '@tanstack/ai'
35
59
 
@@ -41,6 +65,75 @@ function isRecord(value: unknown): value is Record<string, unknown> {
41
65
  return typeof value === 'object' && value !== null
42
66
  }
43
67
 
68
+ function readURLCitation(
69
+ value: unknown,
70
+ ): ResponseOutputText.URLCitation | undefined {
71
+ if (!isRecord(value) || value.type !== 'url_citation') return undefined
72
+ if (
73
+ typeof value.url !== 'string' ||
74
+ typeof value.title !== 'string' ||
75
+ typeof value.start_index !== 'number' ||
76
+ typeof value.end_index !== 'number'
77
+ ) {
78
+ return undefined
79
+ }
80
+ return {
81
+ type: 'url_citation',
82
+ url: value.url,
83
+ title: value.title,
84
+ start_index: value.start_index,
85
+ end_index: value.end_index,
86
+ }
87
+ }
88
+
89
+ function readWebSearchCall(
90
+ value: unknown,
91
+ ): ResponseFunctionWebSearch | undefined {
92
+ if (!isRecord(value) || value.type !== 'web_search_call') return undefined
93
+ if (
94
+ typeof value.id !== 'string' ||
95
+ typeof value.status !== 'string' ||
96
+ !isRecord(value.action) ||
97
+ typeof value.action.type !== 'string'
98
+ ) {
99
+ return undefined
100
+ }
101
+ // oxlint-disable-next-line eslint-js/no-restricted-syntax -- the runtime guard above validates the stable fields while the SDK union preserves provider-specific action fields
102
+ return value as unknown as ResponseFunctionWebSearch
103
+ }
104
+
105
+ function collectWebSearchSources(
106
+ item: ResponseFunctionWebSearch,
107
+ citations: ReadonlyArray<ResponseOutputText.URLCitation>,
108
+ ): Array<ProviderExecutedToolSource> {
109
+ const sources = new Map<string, ProviderExecutedToolSource>()
110
+ const add = (url: unknown, title?: unknown) => {
111
+ if (typeof url !== 'string' || url.length === 0) return
112
+ const existing = sources.get(url)
113
+ if (existing) {
114
+ if (!existing.title && typeof title === 'string' && title.length > 0) {
115
+ existing.title = title
116
+ }
117
+ return
118
+ }
119
+ sources.set(url, {
120
+ url,
121
+ ...(typeof title === 'string' && title.length > 0 ? { title } : {}),
122
+ })
123
+ }
124
+
125
+ if (item.action.type === 'search') {
126
+ for (const source of item.action.sources ?? []) {
127
+ add(source.url)
128
+ }
129
+ }
130
+ for (const citation of citations) {
131
+ add(citation.url, citation.title)
132
+ }
133
+
134
+ return [...sources.values()]
135
+ }
136
+
44
137
  function packResponsesReasoningSignature(
45
138
  id: string | undefined,
46
139
  encryptedContent: string | undefined,
@@ -97,8 +190,17 @@ function readReasoningItem(
97
190
  * TanStack AI uses `call_id` as the canonical tool-call ID and carries the
98
191
  * item ID here so stateless follow-up requests can replay both values.
99
192
  */
100
- export interface OpenAIResponsesToolCallMetadata {
101
- itemId: string
193
+ export interface OpenAIResponsesToolCallMetadata extends ProviderExecutedToolMetadata {
194
+ itemId?: string
195
+ /** Set for shell, local_shell, and apply_patch calls the app must run. */
196
+ openaiUserTool?: OpenAIUserToolName
197
+ /** Shell `action.max_output_length`, echoed on `shell_call_output`. */
198
+ maxOutputLength?: number | null
199
+ openai?: {
200
+ webSearchCall: ResponseFunctionWebSearch
201
+ urlCitations: Array<ResponseOutputText.URLCitation>
202
+ assistantMessage?: ResponseOutputMessage
203
+ }
102
204
  }
103
205
 
104
206
  interface StreamedFunctionCallMetadata {
@@ -145,10 +247,19 @@ export abstract class OpenAIBaseResponsesTextAdapter<
145
247
  readonly name: string
146
248
  protected client: OpenAI
147
249
 
148
- constructor(model: TModel, name: string, client: OpenAI) {
250
+ /** See {@link OpenAIBaseTextAdapterOptions.strictFallbackWarning}. */
251
+ protected readonly strictFallbackWarning: boolean
252
+
253
+ constructor(
254
+ model: TModel,
255
+ name: string,
256
+ client: OpenAI,
257
+ options: OpenAIBaseTextAdapterOptions = {},
258
+ ) {
149
259
  super({}, model)
150
260
  this.name = name
151
261
  this.client = client
262
+ this.strictFallbackWarning = options.strictFallbackWarning ?? true
152
263
  }
153
264
 
154
265
  async *chatStream(
@@ -387,6 +498,7 @@ export abstract class OpenAIBaseResponsesTextAdapter<
387
498
  let hasClosedReasoning = false
388
499
  let model: string = chatOptions.model
389
500
  let usage: OpenAI.Responses.Response['usage'] | undefined
501
+ let responseCompleted = false
390
502
 
391
503
  const closeReasoning = function* (this: {
392
504
  name: string
@@ -583,10 +695,12 @@ export abstract class OpenAIBaseResponsesTextAdapter<
583
695
  }
584
696
 
585
697
  if (chunk.type === 'response.completed') {
698
+ responseCompleted = true
586
699
  const response = chunk.response
587
700
  if (response.usage) usage = response.usage
588
701
  if (response.model) model = response.model
589
- continue
702
+ // Terminal event: do not wait for the HTTP body to close (#1445).
703
+ break
590
704
  }
591
705
 
592
706
  if (chunk.type === 'response.failed') {
@@ -624,6 +738,20 @@ export abstract class OpenAIBaseResponsesTextAdapter<
624
738
  }
625
739
  }
626
740
 
741
+ if (!responseCompleted) {
742
+ const message = 'Response stream ended before response.completed'
743
+ yield {
744
+ type: EventType.RUN_ERROR,
745
+ runId: aguiState.runId,
746
+ model,
747
+ timestamp: Date.now(),
748
+ message,
749
+ code: 'incomplete-stream',
750
+ error: { message, code: 'incomplete-stream' },
751
+ }
752
+ return
753
+ }
754
+
627
755
  if (accumulatedContent.length === 0) {
628
756
  yield {
629
757
  type: EventType.RUN_ERROR,
@@ -897,9 +1025,93 @@ export abstract class OpenAIBaseResponsesTextAdapter<
897
1025
  // cuts off without a response.completed event.
898
1026
  let runFinishedEmitted = false
899
1027
 
1028
+ const providerWebSearchCalls = new Map<
1029
+ string,
1030
+ {
1031
+ item: ResponseFunctionWebSearch
1032
+ index: number
1033
+ started: boolean
1034
+ }
1035
+ >()
1036
+ const webSearchCitations: Array<ResponseOutputText.URLCitation> = []
1037
+
900
1038
  const adapterName = this.name
901
1039
  const emitModel = () => model || options.model
902
1040
 
1041
+ const recordProviderWebSearchCall = (value: unknown, index: number) => {
1042
+ const item = readWebSearchCall(value)
1043
+ if (!item) return
1044
+ const existing = providerWebSearchCalls.get(item.id)
1045
+ if (existing) {
1046
+ existing.item = item
1047
+ existing.index = index
1048
+ } else {
1049
+ providerWebSearchCalls.set(item.id, {
1050
+ item,
1051
+ index,
1052
+ started: false,
1053
+ })
1054
+ }
1055
+ }
1056
+
1057
+ const emitProviderWebSearchCalls = function* (
1058
+ assistantMessage?: ResponseOutputMessage,
1059
+ completedOnly = false,
1060
+ ): Generator<AdapterYieldChunk> {
1061
+ for (const entry of providerWebSearchCalls.values()) {
1062
+ if (
1063
+ entry.started ||
1064
+ (completedOnly && entry.item.status !== 'completed')
1065
+ ) {
1066
+ continue
1067
+ }
1068
+
1069
+ // Citations belong to the whole response. Give a call only the
1070
+ // citations whose URL is in its own action.sources.
1071
+ // ponytail: exact URL match; normalize URLs if the two ever differ.
1072
+ const callUrls = new Set(
1073
+ entry.item.action.type === 'search'
1074
+ ? (entry.item.action.sources ?? []).map((source) => source.url)
1075
+ : [],
1076
+ )
1077
+ const citations = webSearchCitations.filter((citation) =>
1078
+ callUrls.has(citation.url),
1079
+ )
1080
+ const metadata: OpenAIResponsesToolCallMetadata = {
1081
+ itemId: entry.item.id,
1082
+ providerExecuted: true,
1083
+ sources: collectWebSearchSources(entry.item, citations),
1084
+ openai: {
1085
+ webSearchCall: entry.item,
1086
+ urlCitations: citations,
1087
+ ...(assistantMessage ? { assistantMessage } : {}),
1088
+ },
1089
+ }
1090
+
1091
+ entry.started = true
1092
+ yield {
1093
+ type: EventType.TOOL_CALL_START,
1094
+ toolCallId: entry.item.id,
1095
+ toolCallName: 'web_search',
1096
+ toolName: 'web_search',
1097
+ parentMessageId: aguiState.messageId,
1098
+ model: emitModel(),
1099
+ timestamp: Date.now(),
1100
+ index: entry.index,
1101
+ metadata,
1102
+ }
1103
+ yield {
1104
+ type: EventType.TOOL_CALL_END,
1105
+ toolCallId: entry.item.id,
1106
+ toolCallName: 'web_search',
1107
+ toolName: 'web_search',
1108
+ model: emitModel(),
1109
+ timestamp: Date.now(),
1110
+ input: entry.item.action,
1111
+ }
1112
+ }
1113
+ }
1114
+
903
1115
  const openReasoning = function* (): Generator<AdapterYieldChunk> {
904
1116
  if (reasoningMessageId) return
905
1117
  reasoningMessageId = generateId(adapterName)
@@ -979,6 +1191,56 @@ export abstract class OpenAIBaseResponsesTextAdapter<
979
1191
  accumulatedReasoning = ''
980
1192
  }
981
1193
 
1194
+ const userToolChunks = (
1195
+ item: unknown,
1196
+ outputIndex: number,
1197
+ bareShell: boolean,
1198
+ ): Array<AdapterYieldChunk> => {
1199
+ const call = readUserExecutedCall(item, { bareShell })
1200
+ if (!call || toolCallMetadata.get(call.callId)?.ended) return []
1201
+ const tracked = toolCallMetadata.get(call.callId) ?? {
1202
+ callId: call.callId,
1203
+ index: outputIndex,
1204
+ name: call.name,
1205
+ started: false,
1206
+ }
1207
+ toolCallMetadata.set(call.callId, tracked)
1208
+ tracked.started = true
1209
+ tracked.ended = true
1210
+ tracked.name = call.name
1211
+ const timestamp = Date.now()
1212
+ const modelName = model || options.model
1213
+ const metadata: OpenAIResponsesToolCallMetadata = {
1214
+ openaiUserTool: call.name,
1215
+ ...(call.itemId ? { itemId: call.itemId } : {}),
1216
+ ...(call.maxOutputLength !== undefined
1217
+ ? { maxOutputLength: call.maxOutputLength }
1218
+ : {}),
1219
+ }
1220
+ return [
1221
+ {
1222
+ type: EventType.TOOL_CALL_START,
1223
+ toolCallId: call.callId,
1224
+ toolCallName: call.name,
1225
+ toolName: call.name,
1226
+ parentMessageId: aguiState.messageId,
1227
+ model: modelName,
1228
+ timestamp,
1229
+ index: outputIndex,
1230
+ metadata,
1231
+ },
1232
+ {
1233
+ type: EventType.TOOL_CALL_END,
1234
+ toolCallId: call.callId,
1235
+ toolCallName: call.name,
1236
+ toolName: call.name,
1237
+ model: modelName,
1238
+ timestamp,
1239
+ input: call.input,
1240
+ },
1241
+ ]
1242
+ }
1243
+
982
1244
  const emitReasoningDelta = function* (
983
1245
  text: string,
984
1246
  ): Generator<AdapterYieldChunk> {
@@ -1084,6 +1346,14 @@ export abstract class OpenAIBaseResponsesTextAdapter<
1084
1346
  chunk.type === 'response.failed' ||
1085
1347
  chunk.type === 'response.incomplete'
1086
1348
  ) {
1349
+ if (chunk.type === 'response.incomplete') {
1350
+ for (const [index, item] of (
1351
+ chunk.response.output ?? []
1352
+ ).entries()) {
1353
+ recordProviderWebSearchCall(item, index)
1354
+ }
1355
+ yield* emitProviderWebSearchCalls(undefined, true)
1356
+ }
1087
1357
  yield* closeReasoning()
1088
1358
  if (hasEmittedTextMessageStart) {
1089
1359
  yield {
@@ -1296,6 +1566,9 @@ export abstract class OpenAIBaseResponsesTextAdapter<
1296
1566
  // handle output_item.added to capture function call metadata (name)
1297
1567
  if (chunk.type === 'response.output_item.added') {
1298
1568
  const item = chunk.item
1569
+ if (item.type === 'web_search_call') {
1570
+ recordProviderWebSearchCall(item, chunk.output_index)
1571
+ }
1299
1572
  if (item.type === 'reasoning') {
1300
1573
  captureReasoningItem(item)
1301
1574
  yield* openReasoning()
@@ -1341,6 +1614,7 @@ export abstract class OpenAIBaseResponsesTextAdapter<
1341
1614
  metadata.started = true
1342
1615
  }
1343
1616
  }
1617
+ yield* userToolChunks(item, chunk.output_index, false)
1344
1618
  }
1345
1619
 
1346
1620
  // Handle function call arguments delta (streaming). Drop the
@@ -1464,6 +1738,9 @@ export abstract class OpenAIBaseResponsesTextAdapter<
1464
1738
  // whose START + END therefore never fired).
1465
1739
  if (chunk.type === 'response.output_item.done') {
1466
1740
  const item = chunk.item
1741
+ if (item.type === 'web_search_call') {
1742
+ recordProviderWebSearchCall(item, chunk.output_index)
1743
+ }
1467
1744
  if (item.type === 'reasoning') {
1468
1745
  captureReasoningItem(item)
1469
1746
  yield* openReasoning()
@@ -1545,14 +1822,32 @@ export abstract class OpenAIBaseResponsesTextAdapter<
1545
1822
  metadata.pendingArguments = undefined
1546
1823
  }
1547
1824
  }
1825
+ yield* userToolChunks(item, chunk.output_index, false)
1826
+ }
1827
+
1828
+ if (chunk.type === 'response.output_text.annotation.added') {
1829
+ const citation = readURLCitation(chunk.annotation)
1830
+ if (citation) webSearchCitations.push(citation)
1548
1831
  }
1549
1832
 
1550
1833
  if (chunk.type === 'response.completed') {
1834
+ const responseOutput = Array.isArray(chunk.response.output)
1835
+ ? chunk.response.output
1836
+ : []
1837
+ const assistantMessage = responseOutput.find(
1838
+ (item): item is ResponseOutputMessage => item.type === 'message',
1839
+ )
1840
+ for (const [index, item] of responseOutput.entries()) {
1841
+ if (item.type === 'web_search_call') {
1842
+ recordProviderWebSearchCall(item, index)
1843
+ }
1844
+ }
1845
+
1551
1846
  // Some Responses API streams, notably reasoning-model responses,
1552
1847
  // can omit text deltas and carry the successful final text only in
1553
1848
  // response.completed.output. Recover that text so consumers never
1554
1849
  // observe an empty result for a successful response.
1555
- const completedText = chunk.response.output
1850
+ const completedText = responseOutput
1556
1851
  .flatMap((item) => (item.type === 'message' ? item.content : []))
1557
1852
  .filter((part) => part.type === 'output_text')
1558
1853
  .map((part) => part.text)
@@ -1582,10 +1877,8 @@ export abstract class OpenAIBaseResponsesTextAdapter<
1582
1877
  }
1583
1878
  }
1584
1879
 
1585
- if (Array.isArray(chunk.response.output)) {
1586
- for (const item of chunk.response.output) {
1587
- captureReasoningItem(item)
1588
- }
1880
+ for (const item of responseOutput) {
1881
+ captureReasoningItem(item)
1589
1882
  }
1590
1883
  // output_text already closed the streamed reasoning item. A second
1591
1884
  // openReasoning() would emit an empty thinking part. Attach the
@@ -1621,11 +1914,11 @@ export abstract class OpenAIBaseResponsesTextAdapter<
1621
1914
  // be silently dropped from the AG-UI stream while `hasFunctionCalls`
1622
1915
  // below still routes the run's finishReason to 'tool_calls' —
1623
1916
  // leaving consumers waiting for tool results they never saw start.
1624
- for (const item of chunk.response.output) {
1917
+ for (const [outputIndex, item] of responseOutput.entries()) {
1625
1918
  if (item.type !== 'function_call' || !item.id) continue
1626
1919
  const metadata = toolCallMetadata.get(item.id) ?? {
1627
1920
  callId: item.call_id || item.id,
1628
- index: 0,
1921
+ index: outputIndex,
1629
1922
  name: item.name || '',
1630
1923
  started: false,
1631
1924
  }
@@ -1697,6 +1990,21 @@ export abstract class OpenAIBaseResponsesTextAdapter<
1697
1990
  }
1698
1991
  }
1699
1992
 
1993
+ yield* emitProviderWebSearchCalls(assistantMessage)
1994
+
1995
+ const shellOutputs = hostedShellCallIds(responseOutput)
1996
+ for (const [outputIndex, item] of responseOutput.entries()) {
1997
+ if (
1998
+ isRecord(item) &&
1999
+ item.type === 'shell_call' &&
2000
+ typeof item.call_id === 'string' &&
2001
+ shellOutputs.has(item.call_id)
2002
+ ) {
2003
+ continue
2004
+ }
2005
+ yield* userToolChunks(item, outputIndex, true)
2006
+ }
2007
+
1700
2008
  yield* closeReasoning()
1701
2009
  // Emit TEXT_MESSAGE_END if we had text content
1702
2010
  if (hasEmittedTextMessageStart) {
@@ -1716,10 +2024,18 @@ export abstract class OpenAIBaseResponsesTextAdapter<
1716
2024
  // The Responses API's incomplete_details.reason ('max_output_tokens'
1717
2025
  // | 'content_filter') maps to the AG-UI finishReason vocabulary:
1718
2026
  // max_output_tokens → 'length', content_filter → 'content_filter'.
1719
- const hasFunctionCalls = chunk.response.output.some(
1720
- (item: unknown) =>
1721
- (item as { type: string }).type === 'function_call',
1722
- )
2027
+ const hasFunctionCalls = responseOutput.some((item) => {
2028
+ if (!isRecord(item)) return false
2029
+ if (item.type === 'function_call') return true
2030
+ if (
2031
+ item.type === 'shell_call' &&
2032
+ typeof item.call_id === 'string' &&
2033
+ shellOutputs.has(item.call_id)
2034
+ ) {
2035
+ return false
2036
+ }
2037
+ return readUserExecutedCall(item, { bareShell: true }) !== null
2038
+ })
1723
2039
  const incompleteReason = chunk.response.incomplete_details?.reason
1724
2040
  const finishReason:
1725
2041
  | 'tool_calls'
@@ -1747,6 +2063,8 @@ export abstract class OpenAIBaseResponsesTextAdapter<
1747
2063
  finishReason,
1748
2064
  }
1749
2065
  runFinishedEmitted = true
2066
+ // Terminal event: do not wait for the HTTP body to close (#1445).
2067
+ return
1750
2068
  }
1751
2069
 
1752
2070
  if (chunk.type === 'error') {
@@ -1775,11 +2093,11 @@ export abstract class OpenAIBaseResponsesTextAdapter<
1775
2093
  }
1776
2094
  }
1777
2095
 
1778
- // Synthetic terminal RUN_FINISHED if the stream ended without a
1779
- // response.completed event (e.g. truncated upstream connection). This
1780
- // mirrors the chat-completions adapter's behavior so consumers always
1781
- // see a terminal event for every started run.
2096
+ // The stream ended without a terminal event (e.g. a truncated
2097
+ // connection). Completion was never confirmed, so this is not a
2098
+ // successful stop (#1447). The partial text was already emitted.
1782
2099
  if (!runFinishedEmitted && aguiState.hasEmittedRunStarted) {
2100
+ yield* emitProviderWebSearchCalls(undefined, true)
1783
2101
  yield* closeReasoning()
1784
2102
  if (hasEmittedTextMessageStart) {
1785
2103
  yield {
@@ -1789,17 +2107,14 @@ export abstract class OpenAIBaseResponsesTextAdapter<
1789
2107
  timestamp: Date.now(),
1790
2108
  }
1791
2109
  }
1792
- // Omit `usage` entirely (vs `usage: undefined`) — the synthetic
1793
- // RUN_FINISHED for truncated streams has no usage data, and AG-UI's
1794
- // `RunFinishedEvent.usage` is optional without `| undefined` under
1795
- // `exactOptionalPropertyTypes`.
2110
+ const message = 'Response stream ended before response.completed'
1796
2111
  yield {
1797
- type: EventType.RUN_FINISHED,
1798
- runId: aguiState.runId,
1799
- threadId: aguiState.threadId,
2112
+ type: EventType.RUN_ERROR,
1800
2113
  model: model || options.model,
1801
2114
  timestamp: Date.now(),
1802
- finishReason: toolCallMetadata.size > 0 ? 'tool_calls' : 'stop',
2115
+ message,
2116
+ code: 'incomplete-stream',
2117
+ error: { message, code: 'incomplete-stream' },
1803
2118
  }
1804
2119
  }
1805
2120
  } catch (error: unknown) {
@@ -1841,6 +2156,9 @@ export abstract class OpenAIBaseResponsesTextAdapter<
1841
2156
  ): Omit<ResponseCreateParams, 'stream'> {
1842
2157
  const input = this.convertMessagesToInput(options.messages)
1843
2158
 
2159
+ if (this.strictFallbackWarning) {
2160
+ warnStrictFallback(options.tools, options.logger)
2161
+ }
1844
2162
  const tools = options.tools
1845
2163
  ? convertToolsToResponsesFormat(
1846
2164
  options.tools,
@@ -1923,10 +2241,33 @@ export abstract class OpenAIBaseResponsesTextAdapter<
1923
2241
  messages: Array<ModelMessage>,
1924
2242
  ): ResponseInput {
1925
2243
  const result: ResponseInput = []
2244
+ // A reasoning item id may only appear once in `input`; replaying the same
2245
+ // id twice fails with "Duplicate item found with id rs_...". Providers have
2246
+ // been observed minting one id for two separate reasoning items, and a
2247
+ // stored transcript can end up carrying it on two assistant messages.
2248
+ const seenReasoningIds = new Set<string>()
2249
+ const userToolCalls = new Map<
2250
+ string,
2251
+ {
2252
+ id: string
2253
+ function: { name: string; arguments: string }
2254
+ metadata?: unknown
2255
+ }
2256
+ >()
1926
2257
 
1927
2258
  for (const message of messages) {
1928
2259
  // Handle tool messages - convert to FunctionToolCallOutput
1929
2260
  if (message.role === 'tool') {
2261
+ const owner = message.toolCallId
2262
+ ? userToolCalls.get(message.toolCallId)
2263
+ : undefined
2264
+ const userOutput = owner
2265
+ ? userToolResultItem(owner, message.content)
2266
+ : null
2267
+ if (userOutput) {
2268
+ result.push(userOutput)
2269
+ continue
2270
+ }
1930
2271
  const toolContent = message.content
1931
2272
  const output: string | Array<ResponseFunctionCallOutputItem> =
1932
2273
  Array.isArray(toolContent)
@@ -1944,11 +2285,21 @@ export abstract class OpenAIBaseResponsesTextAdapter<
1944
2285
 
1945
2286
  // Handle assistant messages
1946
2287
  if (message.role === 'assistant') {
2288
+ // Every reasoning item this message could contribute, and how many of
2289
+ // them actually made it into `input` after de-duplication. Both counts
2290
+ // are needed below to decide whether the function calls can still be
2291
+ // paired with their reasoning.
2292
+ let reasoningCandidates = 0
2293
+ let emittedReasoning = 0
1947
2294
  if (message.thinking) {
1948
2295
  for (const thinking of message.thinking) {
1949
2296
  if (!thinking.signature) continue
1950
2297
  const packed = unpackResponsesReasoningSignature(thinking.signature)
1951
2298
  if (!packed?.id) continue
2299
+ reasoningCandidates++
2300
+ if (seenReasoningIds.has(packed.id)) continue
2301
+ seenReasoningIds.add(packed.id)
2302
+ emittedReasoning++
1952
2303
  result.push({
1953
2304
  type: 'reasoning',
1954
2305
  id: packed.id,
@@ -1962,31 +2313,83 @@ export abstract class OpenAIBaseResponsesTextAdapter<
1962
2313
  }
1963
2314
  }
1964
2315
 
2316
+ // The Responses API requires a persisted `function_call` item to sit
2317
+ // directly after the reasoning item that produced it. A ModelMessage
2318
+ // stores reasoning and tool calls as two flat arrays, so their original
2319
+ // interleaving is gone: a message holding two or more reasoning items
2320
+ // replays as reasoning A, reasoning B, call A, call B, and the request
2321
+ // is rejected with "Item 'fc_...' of type 'function_call' was provided
2322
+ // without its required 'reasoning' item: 'rs_...'". The same happens
2323
+ // when this message's reasoning was dropped above as a duplicate.
2324
+ //
2325
+ // Sending the calls without their item id makes them fresh items that
2326
+ // the API does not try to pair, which replays correctly. `call_id` is
2327
+ // untouched, so the matching `function_call_output` still resolves.
2328
+ // One reasoning item (or none at all) always lands adjacent to its
2329
+ // calls, so those keep their ids and their prompt-cache hits. If any
2330
+ // reasoning was dropped as a duplicate, a call may belong to it, so
2331
+ // the ids go too.
2332
+ const canPairReasoning =
2333
+ reasoningCandidates === 0 ||
2334
+ (reasoningCandidates === 1 && emittedReasoning === 1)
2335
+
1965
2336
  // If the assistant message has tool calls, add them as FunctionToolCall objects
1966
2337
  // Responses API expects arguments as a string (JSON string)
2338
+ let rawAssistantMessage: ResponseOutputMessage | undefined
1967
2339
  if (message.toolCalls && message.toolCalls.length > 0) {
1968
2340
  for (const toolCall of message.toolCalls) {
2341
+ const metadata = toolCall.metadata as
2342
+ | OpenAIResponsesToolCallMetadata
2343
+ | undefined
2344
+ if (metadata?.providerExecuted) {
2345
+ const webSearchCall = metadata.openai?.webSearchCall
2346
+ // The raw ws_/msg_ items carry ids that must pair with their
2347
+ // reasoning item, the same as function calls above. When they
2348
+ // cannot pair, skip them and send the plain message with no id.
2349
+ if (webSearchCall && canPairReasoning) {
2350
+ result.push(webSearchCall)
2351
+ rawAssistantMessage ??= metadata.openai?.assistantMessage
2352
+ }
2353
+ continue
2354
+ }
2355
+
1969
2356
  // Keep arguments as string for Responses API
1970
2357
  const argumentsString =
1971
2358
  typeof toolCall.function.arguments === 'string'
1972
2359
  ? toolCall.function.arguments
1973
2360
  : JSON.stringify(toolCall.function.arguments)
1974
- const itemId = (
1975
- toolCall.metadata as OpenAIResponsesToolCallMetadata | undefined
1976
- )?.itemId
2361
+ const replayCall = {
2362
+ id: toolCall.id,
2363
+ function: {
2364
+ name: toolCall.function.name,
2365
+ arguments: argumentsString,
2366
+ },
2367
+ ...(toolCall.metadata !== undefined
2368
+ ? { metadata: toolCall.metadata }
2369
+ : {}),
2370
+ }
2371
+ const userItem = userToolRequestItem(replayCall)
2372
+ if (userItem) {
2373
+ userToolCalls.set(toolCall.id, replayCall)
2374
+ result.push(userItem)
2375
+ continue
2376
+ }
2377
+ const itemId = metadata?.itemId
1977
2378
 
1978
2379
  result.push({
1979
2380
  type: 'function_call',
1980
2381
  call_id: toolCall.id,
1981
- ...(itemId && { id: itemId }),
2382
+ ...(itemId && canPairReasoning && { id: itemId }),
1982
2383
  name: toolCall.function.name,
1983
2384
  arguments: argumentsString,
1984
2385
  })
1985
2386
  }
1986
2387
  }
1987
2388
 
1988
- // Add the assistant's text message if there is content
1989
- if (message.content) {
2389
+ if (rawAssistantMessage) {
2390
+ result.push(rawAssistantMessage)
2391
+ } else if (message.content) {
2392
+ // Add the assistant's text message if there is content
1990
2393
  const contentStr = this.extractTextContent(message.content)
1991
2394
  if (contentStr) {
1992
2395
  result.push({
@@ -2046,6 +2449,16 @@ export abstract class OpenAIBaseResponsesTextAdapter<
2046
2449
  const imageMetadata = part.metadata as
2047
2450
  | { detail?: 'auto' | 'low' | 'high' }
2048
2451
  | undefined
2452
+ if (isFileSource(part.source)) {
2453
+ if (this.supportsFileSources !== true) {
2454
+ throw unsupportedFileSourceError(this.name)
2455
+ }
2456
+ return {
2457
+ type: 'input_image',
2458
+ file_id: fileReferenceFor(part.source, this.name),
2459
+ detail: imageMetadata?.detail || 'auto',
2460
+ }
2461
+ }
2049
2462
  if (part.source.type === 'url') {
2050
2463
  return {
2051
2464
  type: 'input_image',
@@ -2069,6 +2482,15 @@ export abstract class OpenAIBaseResponsesTextAdapter<
2069
2482
  }
2070
2483
  }
2071
2484
  case 'audio': {
2485
+ if (isFileSource(part.source)) {
2486
+ if (this.supportsFileSources !== true) {
2487
+ throw unsupportedFileSourceError(this.name)
2488
+ }
2489
+ return {
2490
+ type: 'input_file',
2491
+ file_id: fileReferenceFor(part.source, this.name),
2492
+ }
2493
+ }
2072
2494
  if (part.source.type === 'url') {
2073
2495
  return {
2074
2496
  type: 'input_file',
@@ -2102,6 +2524,16 @@ export abstract class OpenAIBaseResponsesTextAdapter<
2102
2524
  documentMetadata?.detail !== undefined
2103
2525
  ? { detail: documentMetadata.detail as 'low' | 'high' }
2104
2526
  : {}
2527
+ if (isFileSource(part.source)) {
2528
+ if (this.supportsFileSources !== true) {
2529
+ throw unsupportedFileSourceError(this.name)
2530
+ }
2531
+ return {
2532
+ type: 'input_file',
2533
+ file_id: fileReferenceFor(part.source, this.name),
2534
+ ...documentDetail,
2535
+ }
2536
+ }
2105
2537
  if (part.source.type === 'url') {
2106
2538
  // The Responses API fetches the PDF itself; filename and MIME
2107
2539
  // type are inferred from the response.
@@ -2164,7 +2596,6 @@ export abstract class OpenAIBaseResponsesTextAdapter<
2164
2596
  ...documentDetail,
2165
2597
  }
2166
2598
  }
2167
-
2168
2599
  case 'video':
2169
2600
  default:
2170
2601
  // OpenAI Responses API doesn't accept native video parts on this