@tanstack/ai-gemini 0.17.3 → 0.18.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,5 +1,6 @@
1
1
  import { EventType } from '@tanstack/ai'
2
2
  import { BaseTextAdapter } from '@tanstack/ai/adapters'
3
+ import { parse as parsePartialJSON } from 'partial-json'
3
4
  import {
4
5
  createGeminiClient,
5
6
  generateId,
@@ -377,10 +378,17 @@ export class GeminiTextInteractionsAdapter<
377
378
  },
378
379
  })
379
380
 
381
+ // SDK 2.x: `response_mime_type` has been removed and `response_format`
382
+ // is now polymorphic — each entry has a `type` discriminator and the
383
+ // mime type lives inside the entry. See:
384
+ // https://ai.google.dev/gemini-api/docs/interactions-breaking-changes-may-2026
380
385
  const request: GeminiInteractionsRequestBody = {
381
386
  ...baseRequest,
382
- response_mime_type: 'application/json',
383
- response_format: outputSchema,
387
+ response_format: {
388
+ type: 'text',
389
+ mime_type: 'application/json',
390
+ schema: outputSchema,
391
+ },
384
392
  }
385
393
 
386
394
  try {
@@ -490,7 +498,6 @@ function buildInteractionsRequest(
490
498
  background: modelOpts?.background,
491
499
  response_modalities: modelOpts?.response_modalities,
492
500
  response_format: modelOpts?.response_format,
493
- response_mime_type: modelOpts?.response_mime_type,
494
501
  }
495
502
  }
496
503
 
@@ -638,24 +645,24 @@ function messagesAfterLastAssistant(
638
645
  return messages
639
646
  }
640
647
 
641
- function safeParseToolArguments(
642
- raw: string | undefined,
643
- logger: InternalLogger,
644
- ): Record<string, unknown> {
645
- if (!raw) return {}
648
+ // Leniently parse the *accumulated* streamed tool-call argument buffer.
649
+ // Streamed `arguments_delta` fragments are individually incomplete JSON, so
650
+ // a strict `JSON.parse` would throw (and log noise) on every fragment until
651
+ // the final one. `partial-json` recovers a best-effort object from a
652
+ // truncated buffer instead. Returns `undefined` when nothing usable could be
653
+ // parsed, so callers can keep the last good value rather than clobber
654
+ // previously-merged args with `{}`.
655
+ function parsePartialToolArguments(
656
+ raw: string,
657
+ ): Record<string, unknown> | undefined {
658
+ if (!raw) return undefined
646
659
  try {
647
- const parsed = JSON.parse(raw)
648
- return parsed && typeof parsed === 'object' ? parsed : {}
649
- } catch (error) {
650
- logger.errors(
651
- 'gemini-text-interactions.safeParseToolArguments parse failed',
652
- {
653
- error,
654
- raw,
655
- source: 'gemini-text-interactions.chatStream',
656
- },
657
- )
658
- return {}
660
+ const parsed = parsePartialJSON(raw)
661
+ return parsed && typeof parsed === 'object' && !Array.isArray(parsed)
662
+ ? (parsed as Record<string, unknown>)
663
+ : undefined
664
+ } catch {
665
+ return undefined
659
666
  }
660
667
  }
661
668
 
@@ -924,6 +931,17 @@ async function* translateInteractionEvents(
924
931
  let thinkingAccumulated = ''
925
932
  let reasoningMessageId: string | null = null
926
933
  let hasClosedReasoning = false
934
+ // SDK 2.x routes events by step `index`, not by content id. We need to
935
+ // map the index of an in-flight `function_call` step back to the tool
936
+ // call id so subsequent `step.delta` (arguments_delta) and `step.stop`
937
+ // events can update / close the right TOOL_CALL_*.
938
+ const indexToToolCallId = new Map<number, string>()
939
+ // Function-call arguments now stream as partial JSON fragments
940
+ // (`StepDelta.ArgumentsDelta.arguments`) rather than as pre-parsed
941
+ // object deltas. Buffer the raw strings per tool call so we can
942
+ // attempt one JSON parse per delta and recover gracefully if the
943
+ // fragment isn't yet syntactically complete.
944
+ const argStringByToolCallId = new Map<string, string>()
927
945
 
928
946
  const closeReasoningIfNeeded = function* (): Generator<StreamChunk> {
929
947
  if (reasoningMessageId && !hasClosedReasoning) {
@@ -997,86 +1015,146 @@ async function* translateInteractionEvents(
997
1015
  for await (const event of stream) {
998
1016
  logger.provider(`provider=gemini-text-interactions`, { event })
999
1017
  switch (event.event_type) {
1000
- case 'interaction.start': {
1018
+ case 'interaction.created': {
1001
1019
  interactionId = event.interaction.id
1002
1020
  yield* emitRunStartedIfNeeded()
1003
1021
  break
1004
1022
  }
1005
1023
 
1006
- case 'content.start': {
1007
- yield* emitRunStartedIfNeeded()
1008
- break
1009
- }
1010
-
1011
- case 'content.delta': {
1024
+ case 'step.start': {
1012
1025
  yield* emitRunStartedIfNeeded()
1013
- const delta = event.delta
1014
- switch (delta.type) {
1015
- case 'text': {
1026
+ const step = event.step
1027
+ const index = event.index
1028
+ switch (step.type) {
1029
+ case 'function_call': {
1016
1030
  yield* closeReasoningIfNeeded()
1017
- if (!hasEmittedTextMessageStart) {
1018
- hasEmittedTextMessageStart = true
1019
- yield {
1020
- type: EventType.TEXT_MESSAGE_START,
1021
- messageId,
1022
- model,
1023
- timestamp,
1024
- role: 'assistant',
1025
- }
1031
+ sawFunctionCall = true
1032
+ const toolCallId = step.id
1033
+ indexToToolCallId.set(index, toolCallId)
1034
+ // `step.arguments` is required on FunctionCallStep but may
1035
+ // be an empty `{}` placeholder when streaming, where the
1036
+ // real args arrive as `arguments_delta` events. Treat both
1037
+ // uniformly: stash whatever we got, stringify once.
1038
+ const initialArgs = step.arguments
1039
+ const state: ToolCallState = {
1040
+ name: step.name,
1041
+ args: { ...initialArgs },
1042
+ index: nextToolIndex++,
1043
+ started: true,
1044
+ ended: false,
1026
1045
  }
1027
- textAccumulated += delta.text
1046
+ toolCalls.set(toolCallId, state)
1047
+ argStringByToolCallId.set(
1048
+ toolCallId,
1049
+ Object.keys(initialArgs).length > 0
1050
+ ? JSON.stringify(initialArgs)
1051
+ : '',
1052
+ )
1028
1053
  yield {
1029
- type: EventType.TEXT_MESSAGE_CONTENT,
1030
- messageId,
1054
+ type: EventType.TOOL_CALL_START,
1055
+ toolCallId,
1056
+ toolCallName: state.name,
1057
+ toolName: state.name,
1058
+ // Bind the tool call to the same assistant message id the
1059
+ // eventual TEXT_MESSAGE_START uses so the message id stays
1060
+ // stable when a function_call arrives before any text (#477).
1061
+ parentMessageId: messageId,
1031
1062
  model,
1032
1063
  timestamp,
1033
- delta: delta.text,
1034
- content: textAccumulated,
1064
+ index: state.index,
1065
+ }
1066
+ if (Object.keys(initialArgs).length > 0) {
1067
+ const argsJson = JSON.stringify(initialArgs)
1068
+ yield {
1069
+ type: EventType.TOOL_CALL_ARGS,
1070
+ toolCallId,
1071
+ model,
1072
+ timestamp,
1073
+ delta: argsJson,
1074
+ args: argsJson,
1075
+ }
1035
1076
  }
1036
1077
  break
1037
1078
  }
1038
- case 'function_call': {
1039
- yield* closeReasoningIfNeeded()
1040
- sawFunctionCall = true
1041
- const toolCallId = delta.id
1042
- const deltaArgs: Record<string, unknown> =
1043
- typeof delta.arguments === 'string'
1044
- ? safeParseToolArguments(delta.arguments, logger)
1045
- : delta.arguments
1046
- let state = toolCalls.get(toolCallId)
1047
- if (!state) {
1048
- state = {
1049
- name: delta.name,
1050
- args: { ...deltaArgs },
1051
- index: nextToolIndex++,
1052
- started: false,
1053
- ended: false,
1079
+ case 'thought': {
1080
+ // Open the reasoning block lazily — content lands here via
1081
+ // `step.delta { thought_summary }` events. If the server
1082
+ // ships a non-empty `summary` array up-front (rare, unary
1083
+ // responses), surface it immediately.
1084
+ if (thinkingStepId === null || reasoningMessageId === null) {
1085
+ thinkingStepId = generateId(adapterName)
1086
+ reasoningMessageId = generateId(adapterName)
1087
+ yield {
1088
+ type: EventType.REASONING_START,
1089
+ messageId: reasoningMessageId,
1090
+ model,
1091
+ timestamp,
1092
+ }
1093
+ yield {
1094
+ type: EventType.REASONING_MESSAGE_START,
1095
+ messageId: reasoningMessageId,
1096
+ role: 'reasoning',
1097
+ model,
1098
+ timestamp,
1099
+ }
1100
+ yield {
1101
+ type: EventType.STEP_STARTED,
1102
+ stepName: thinkingStepId,
1103
+ stepId: thinkingStepId,
1104
+ model,
1105
+ timestamp,
1106
+ stepType: 'thinking',
1054
1107
  }
1055
- toolCalls.set(toolCallId, state)
1056
- } else {
1057
- state.args = { ...state.args, ...deltaArgs }
1058
- if (delta.name) state.name = delta.name
1059
1108
  }
1060
- if (!state.started) {
1061
- state.started = true
1109
+ for (const part of step.summary ?? []) {
1110
+ if (part.type !== 'text' || !part.text) continue
1111
+ thinkingAccumulated += part.text
1062
1112
  yield {
1063
- type: EventType.TOOL_CALL_START,
1064
- toolCallId,
1065
- toolCallName: state.name,
1066
- toolName: state.name,
1067
- parentMessageId: messageId,
1113
+ type: EventType.REASONING_MESSAGE_CONTENT,
1114
+ messageId: reasoningMessageId,
1115
+ delta: part.text,
1068
1116
  model,
1069
1117
  timestamp,
1070
- index: state.index,
1118
+ }
1119
+ yield {
1120
+ type: EventType.STEP_FINISHED,
1121
+ stepName: thinkingStepId,
1122
+ stepId: thinkingStepId,
1123
+ model,
1124
+ timestamp,
1125
+ delta: part.text,
1126
+ content: thinkingAccumulated,
1071
1127
  }
1072
1128
  }
1073
- yield {
1074
- type: EventType.TOOL_CALL_ARGS,
1075
- toolCallId,
1076
- model,
1077
- timestamp,
1078
- delta: JSON.stringify(deltaArgs),
1079
- args: JSON.stringify(state.args),
1129
+ break
1130
+ }
1131
+ case 'model_output': {
1132
+ yield* closeReasoningIfNeeded()
1133
+ // Some servers ship an initial `content` array on
1134
+ // `step.start` (notably non-streaming and ahead-of-stream
1135
+ // unary completions). Treat any prefilled text content the
1136
+ // same way a `text` step.delta would.
1137
+ for (const part of step.content ?? []) {
1138
+ if (part.type !== 'text' || !part.text) continue
1139
+ if (!hasEmittedTextMessageStart) {
1140
+ hasEmittedTextMessageStart = true
1141
+ yield {
1142
+ type: EventType.TEXT_MESSAGE_START,
1143
+ messageId,
1144
+ model,
1145
+ timestamp,
1146
+ role: 'assistant',
1147
+ }
1148
+ }
1149
+ textAccumulated += part.text
1150
+ yield {
1151
+ type: EventType.TEXT_MESSAGE_CONTENT,
1152
+ messageId,
1153
+ model,
1154
+ timestamp,
1155
+ delta: part.text,
1156
+ content: textAccumulated,
1157
+ }
1080
1158
  }
1081
1159
  break
1082
1160
  }
@@ -1085,7 +1163,7 @@ async function* translateInteractionEvents(
1085
1163
  yield {
1086
1164
  type: EventType.CUSTOM,
1087
1165
  name: 'gemini.googleSearchCall',
1088
- value: delta,
1166
+ value: step,
1089
1167
  model,
1090
1168
  timestamp,
1091
1169
  }
@@ -1096,7 +1174,7 @@ async function* translateInteractionEvents(
1096
1174
  yield {
1097
1175
  type: EventType.CUSTOM,
1098
1176
  name: 'gemini.googleSearchResult',
1099
- value: delta,
1177
+ value: step,
1100
1178
  model,
1101
1179
  timestamp,
1102
1180
  }
@@ -1107,7 +1185,7 @@ async function* translateInteractionEvents(
1107
1185
  yield {
1108
1186
  type: EventType.CUSTOM,
1109
1187
  name: 'gemini.codeExecutionCall',
1110
- value: delta,
1188
+ value: step,
1111
1189
  model,
1112
1190
  timestamp,
1113
1191
  }
@@ -1118,7 +1196,7 @@ async function* translateInteractionEvents(
1118
1196
  yield {
1119
1197
  type: EventType.CUSTOM,
1120
1198
  name: 'gemini.codeExecutionResult',
1121
- value: delta,
1199
+ value: step,
1122
1200
  model,
1123
1201
  timestamp,
1124
1202
  }
@@ -1129,7 +1207,7 @@ async function* translateInteractionEvents(
1129
1207
  yield {
1130
1208
  type: EventType.CUSTOM,
1131
1209
  name: 'gemini.urlContextCall',
1132
- value: delta,
1210
+ value: step,
1133
1211
  model,
1134
1212
  timestamp,
1135
1213
  }
@@ -1140,7 +1218,7 @@ async function* translateInteractionEvents(
1140
1218
  yield {
1141
1219
  type: EventType.CUSTOM,
1142
1220
  name: 'gemini.urlContextResult',
1143
- value: delta,
1221
+ value: step,
1144
1222
  model,
1145
1223
  timestamp,
1146
1224
  }
@@ -1151,7 +1229,7 @@ async function* translateInteractionEvents(
1151
1229
  yield {
1152
1230
  type: EventType.CUSTOM,
1153
1231
  name: 'gemini.fileSearchCall',
1154
- value: delta,
1232
+ value: step,
1155
1233
  model,
1156
1234
  timestamp,
1157
1235
  }
@@ -1162,12 +1240,94 @@ async function* translateInteractionEvents(
1162
1240
  yield {
1163
1241
  type: EventType.CUSTOM,
1164
1242
  name: 'gemini.fileSearchResult',
1165
- value: delta,
1243
+ value: step,
1166
1244
  model,
1167
1245
  timestamp,
1168
1246
  }
1169
1247
  break
1170
1248
  }
1249
+ // Unhandled step types (user_input on GET timelines,
1250
+ // mcp_server_*, google_maps_*, function_result) fall through
1251
+ // to the observability default so SDK drift is visible.
1252
+ case 'user_input':
1253
+ case 'mcp_server_tool_call':
1254
+ case 'mcp_server_tool_result':
1255
+ case 'google_maps_call':
1256
+ case 'google_maps_result':
1257
+ case 'function_result':
1258
+ default:
1259
+ logger.provider(`gemini-text-interactions unhandled step.start`, {
1260
+ step,
1261
+ })
1262
+ break
1263
+ }
1264
+ break
1265
+ }
1266
+
1267
+ case 'step.delta': {
1268
+ yield* emitRunStartedIfNeeded()
1269
+ const delta = event.delta
1270
+ const index = event.index
1271
+ switch (delta.type) {
1272
+ case 'text': {
1273
+ yield* closeReasoningIfNeeded()
1274
+ if (!hasEmittedTextMessageStart) {
1275
+ hasEmittedTextMessageStart = true
1276
+ yield {
1277
+ type: EventType.TEXT_MESSAGE_START,
1278
+ messageId,
1279
+ model,
1280
+ timestamp,
1281
+ role: 'assistant',
1282
+ }
1283
+ }
1284
+ textAccumulated += delta.text
1285
+ yield {
1286
+ type: EventType.TEXT_MESSAGE_CONTENT,
1287
+ messageId,
1288
+ model,
1289
+ timestamp,
1290
+ delta: delta.text,
1291
+ content: textAccumulated,
1292
+ }
1293
+ break
1294
+ }
1295
+ case 'arguments_delta': {
1296
+ // Streamed function-call arguments. Identity (id, name) was
1297
+ // delivered on the matching `step.start` and recorded in
1298
+ // indexToToolCallId.
1299
+ const toolCallId = indexToToolCallId.get(index)
1300
+ if (!toolCallId) {
1301
+ logger.provider(
1302
+ `gemini-text-interactions arguments_delta for unknown step index`,
1303
+ { index, delta },
1304
+ )
1305
+ break
1306
+ }
1307
+ const state = toolCalls.get(toolCallId)
1308
+ if (!state) break
1309
+ const fragment = delta.arguments ?? ''
1310
+ const buffer =
1311
+ (argStringByToolCallId.get(toolCallId) ?? '') + fragment
1312
+ argStringByToolCallId.set(toolCallId, buffer)
1313
+ // Parse the accumulated buffer leniently: streamed arg fragments
1314
+ // are individually incomplete JSON, so use a partial-JSON parser
1315
+ // that tolerates truncation rather than logging a parse error per
1316
+ // fragment. Only overwrite `state.args` when we actually recovered
1317
+ // an object, so a momentarily-unparseable fragment can't reset
1318
+ // previously-merged args back to `{}`.
1319
+ const parsed = parsePartialToolArguments(buffer)
1320
+ if (parsed) state.args = parsed
1321
+ yield {
1322
+ type: EventType.TOOL_CALL_ARGS,
1323
+ toolCallId,
1324
+ model,
1325
+ timestamp,
1326
+ delta: fragment,
1327
+ args: buffer,
1328
+ }
1329
+ break
1330
+ }
1171
1331
  case 'thought_summary': {
1172
1332
  const thoughtText =
1173
1333
  delta.content && 'text' in delta.content ? delta.content.text : ''
@@ -1216,22 +1376,33 @@ async function* translateInteractionEvents(
1216
1376
  }
1217
1377
  break
1218
1378
  }
1219
- // The following delta types are valid per the SDK type union
1220
- // but aren't yet translated by this adapter (output modalities
1221
- // text-only adapter shouldn't see, response-side function_result
1222
- // / mcp_server_*, thought_signature). Falling through to the
1223
- // observability default so SDK drift is visible.
1379
+ // The remaining StepDelta variants (image/audio/video/document
1380
+ // for output modalities a text adapter shouldn't see, tool
1381
+ // call/result deltas which are surfaced via step.start in this
1382
+ // adapter, thought_signature, annotation deltas, mcp/google
1383
+ // maps variants) fall through to the observability default.
1224
1384
  case 'image':
1225
1385
  case 'audio':
1226
1386
  case 'video':
1227
1387
  case 'document':
1228
- case 'function_result':
1388
+ case 'thought_signature':
1389
+ case 'text_annotation_delta':
1390
+ case 'code_execution_call':
1391
+ case 'code_execution_result':
1392
+ case 'url_context_call':
1393
+ case 'url_context_result':
1394
+ case 'google_search_call':
1395
+ case 'google_search_result':
1396
+ case 'file_search_call':
1397
+ case 'file_search_result':
1229
1398
  case 'mcp_server_tool_call':
1230
1399
  case 'mcp_server_tool_result':
1231
- case 'thought_signature':
1400
+ case 'google_maps_call':
1401
+ case 'google_maps_result':
1402
+ case 'function_result':
1232
1403
  default:
1233
1404
  logger.provider(
1234
- `gemini-text-interactions unhandled content.delta type`,
1405
+ `gemini-text-interactions unhandled step.delta type`,
1235
1406
  { delta },
1236
1407
  )
1237
1408
  break
@@ -1239,12 +1410,34 @@ async function* translateInteractionEvents(
1239
1410
  break
1240
1411
  }
1241
1412
 
1242
- case 'content.stop':
1413
+ case 'step.stop': {
1414
+ // Close any open function_call so downstream consumers get the
1415
+ // matching TOOL_CALL_END once the arguments are complete. Other
1416
+ // step types don't carry adapter-level open state.
1417
+ const toolCallId = indexToToolCallId.get(event.index)
1418
+ if (toolCallId) {
1419
+ const state = toolCalls.get(toolCallId)
1420
+ if (state && !state.ended) {
1421
+ state.ended = true
1422
+ yield {
1423
+ type: EventType.TOOL_CALL_END,
1424
+ toolCallId,
1425
+ toolName: state.name,
1426
+ model,
1427
+ timestamp,
1428
+ input: state.args,
1429
+ }
1430
+ }
1431
+ indexToToolCallId.delete(event.index)
1432
+ }
1433
+ break
1434
+ }
1435
+
1243
1436
  case 'interaction.status_update': {
1244
1437
  break
1245
1438
  }
1246
1439
 
1247
- case 'interaction.complete': {
1440
+ case 'interaction.completed': {
1248
1441
  if (event.interaction.id) {
1249
1442
  interactionId = event.interaction.id
1250
1443
  }
@@ -1348,10 +1541,21 @@ async function* translateInteractionEvents(
1348
1541
  }
1349
1542
 
1350
1543
  function extractTextFromInteraction(interaction: Interaction): string {
1544
+ // SDK 2.x: the response carries a `steps` array; `output_text` is a
1545
+ // convenience the SDK derives from the last model output. Prefer the
1546
+ // SDK sugar when it's populated, then fall back to walking
1547
+ // model_output steps for adapters / responses that don't get the
1548
+ // sugar (e.g. older SDK builds).
1549
+ if (typeof interaction.output_text === 'string' && interaction.output_text) {
1550
+ return interaction.output_text
1551
+ }
1351
1552
  let text = ''
1352
- for (const output of interaction.outputs ?? []) {
1353
- if (output.type === 'text') {
1354
- text += output.text
1553
+ for (const step of interaction.steps) {
1554
+ if (step.type !== 'model_output' || !step.content) continue
1555
+ for (const part of step.content) {
1556
+ if (part.type === 'text') {
1557
+ text += part.text
1558
+ }
1355
1559
  }
1356
1560
  }
1357
1561
  return text