@tanstack/ai-gemini 0.11.0 → 0.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1392 @@
1
+ import { EventType } from '@tanstack/ai'
2
+ import { BaseTextAdapter } from '@tanstack/ai/adapters'
3
+ import {
4
+ createGeminiClient,
5
+ generateId,
6
+ getGeminiApiKeyFromEnv,
7
+ } from '../../utils'
8
+ import type { InternalLogger } from '@tanstack/ai/adapter-internals'
9
+ import type {
10
+ GeminiChatModelToolCapabilitiesByName,
11
+ GeminiModelInputModalitiesByName,
12
+ GeminiModels,
13
+ } from '../../model-meta'
14
+ import type {
15
+ StructuredOutputOptions,
16
+ StructuredOutputResult,
17
+ } from '@tanstack/ai/adapters'
18
+ import type { GoogleGenAI, Interactions } from '@google/genai'
19
+ import type {
20
+ ContentPart,
21
+ Modality,
22
+ ModelMessage,
23
+ StreamChunk,
24
+ TextOptions,
25
+ Tool,
26
+ } from '@tanstack/ai'
27
+
28
+ import type {
29
+ GeminiInteractionsCustomEvent,
30
+ GeminiInteractionsCustomEventValue,
31
+ GeminiInteractionsStream,
32
+ } from './events'
33
+ import type { ExternalTextInteractionsProviderOptions } from './provider-options'
34
+ import type { GeminiMessageMetadataByModality } from '../../message-types'
35
+ import type { GeminiClientConfig } from '../../utils'
36
+
37
+ type Interaction = Interactions.Interaction
38
+ type InteractionSSEEvent = Interactions.InteractionSSEEvent
39
+
40
+ export type GeminiTextInteractionsConfig = GeminiClientConfig
41
+
42
+ export type GeminiTextInteractionsProviderOptions =
43
+ ExternalTextInteractionsProviderOptions
44
+
45
+ type InteractionsTool = NonNullable<
46
+ Interactions.CreateModelInteractionParamsStreaming['tools']
47
+ >[number]
48
+
49
+ type ContentBlock = Interactions.Content
50
+
51
+ // The Interactions API takes `input` as a list of *Steps* (not a list of
52
+ // content blocks, and not a list of `Turn`s — the SDK's type union is
53
+ // misleading on both counts). The live API enforces the Step envelope —
54
+ // raw content arrays produce `invalid_request` / "value at top-level
55
+ // must be a list". The wire discriminator is snake_case
56
+ // (`user_input` / `function_result`); see
57
+ // https://ai.google.dev/api/interactions-api for the full Step union.
58
+ type UserInputStep = {
59
+ type: 'user_input'
60
+ content: Array<ContentBlock>
61
+ }
62
+ type FunctionResultStep = {
63
+ type: 'function_result'
64
+ call_id: string
65
+ name?: string
66
+ result: string
67
+ }
68
+ type InteractionsStep = UserInputStep | FunctionResultStep
69
+ type InteractionsRequestInput = Array<InteractionsStep>
70
+
71
+ // Concrete wire shape we send to `client.interactions.create`. The SDK's
72
+ // own param union types `input` as `string | Content[] | Turn[] | ...`
73
+ // which is wrong for the live API (see the InteractionsRequestInput
74
+ // comment above), so we type `input` ourselves and cast just once at the
75
+ // SDK boundary instead of casting every field through.
76
+ type GeminiInteractionsRequestBody = Omit<
77
+ Interactions.CreateModelInteractionParamsStreaming,
78
+ 'input' | 'stream'
79
+ > & {
80
+ input: InteractionsRequestInput
81
+ stream?: boolean
82
+ }
83
+
84
+ type ToolCallState = {
85
+ name: string
86
+ // Accumulated args as a parsed object. Kept here in object form so a
87
+ // garbled delta can't corrupt previously-merged fragments (the prior
88
+ // string-then-reparse pipeline replaced the whole accumulator on any
89
+ // parse failure). Stringified only when emitting AG-UI events.
90
+ args: Record<string, unknown>
91
+ index: number
92
+ started: boolean
93
+ ended: boolean
94
+ }
95
+
96
+ // ===========================
97
+ // Type Resolution Helpers
98
+ // ===========================
99
+
100
+ /**
101
+ * Resolve provider options for a specific model. The Interactions API's
102
+ * request shape is the same across all chat-capable Gemini models — the
103
+ * SDK doesn't expose a per-model param union — so this currently falls
104
+ * through to the flat `GeminiTextInteractionsProviderOptions` for every
105
+ * model. The alias exists for parity with `GeminiTextAdapter`, so a
106
+ * per-model map can be slotted in later without changing the adapter
107
+ * signature.
108
+ */
109
+ type ResolveProviderOptions = GeminiTextInteractionsProviderOptions
110
+
111
+ /**
112
+ * Resolve input modalities for a specific model. Reuses the chat-model
113
+ * modality map from `model-meta.ts`: passing a `document` content block
114
+ * to a model that doesn't support it is a compile error, matching the
115
+ * sibling `GeminiTextAdapter`.
116
+ */
117
+ type ResolveInputModalities<TModel extends string> =
118
+ TModel extends keyof GeminiModelInputModalitiesByName
119
+ ? GeminiModelInputModalitiesByName[TModel]
120
+ : readonly ['text', 'image', 'audio', 'video', 'document']
121
+
122
+ /**
123
+ * Resolve tool capabilities for a specific model. Reuses the chat-model
124
+ * capability map: `google_maps` / `google_search_retrieval` /
125
+ * `mcp_server` are rejected at runtime by `convertToolsToInteractionsFormat`,
126
+ * but per-model gating happens here at compile time.
127
+ */
128
+ type ResolveToolCapabilities<TModel extends string> =
129
+ TModel extends keyof GeminiChatModelToolCapabilitiesByName
130
+ ? NonNullable<GeminiChatModelToolCapabilitiesByName[TModel]>
131
+ : readonly []
132
+
133
+ /**
134
+ * Tree-shakeable adapter for Gemini's stateful Interactions API. Routes
135
+ * through `client.interactions.create` and surfaces the server-assigned
136
+ * `interactionId` via an AG-UI `CUSTOM` event with
137
+ * `name: 'gemini.interactionId'` emitted just before `RUN_FINISHED`; pass
138
+ * that id back on the next turn via `modelOptions.previous_interaction_id`
139
+ * to continue the conversation without resending history.
140
+ *
141
+ * The Interactions API does NOT support stateless multi-turn replay —
142
+ * passing more than one message in `messages` without a
143
+ * `previous_interaction_id` throws. For a chat UI that maintains local
144
+ * history (e.g. `useChat`), see the "Wiring with `useChat`" section of
145
+ * `docs/adapters/gemini.md` for the canonical client/server pattern.
146
+ *
147
+ * Supports user-defined function tools and the built-in tools
148
+ * `google_search`, `code_execution`, `url_context`, `file_search`, and
149
+ * `computer_use`. Built-in tool *activity* for the four search/exec
150
+ * variants is surfaced via `CUSTOM` events
151
+ * (`gemini.googleSearchCall` / `gemini.googleSearchResult` and the
152
+ * corresponding per-tool variants) carrying the raw Interactions delta;
153
+ * see {@link GeminiInteractionsCustomEvent}. `computer_use` is accepted
154
+ * in the request but the Interactions API does not currently stream
155
+ * per-delta CUSTOM events for it. `google_search_retrieval`,
156
+ * `google_maps`, and `mcp_server` are not supported on this adapter.
157
+ *
158
+ * @experimental Interactions API is in Beta per Google; shapes may change.
159
+ * @see https://ai.google.dev/gemini-api/docs/interactions
160
+ */
161
+ export class GeminiTextInteractionsAdapter<
162
+ TModel extends GeminiModels,
163
+ TProviderOptions extends Record<string, any> = ResolveProviderOptions,
164
+ TInputModalities extends ReadonlyArray<Modality> =
165
+ ResolveInputModalities<TModel>,
166
+ TToolCapabilities extends ReadonlyArray<string> =
167
+ ResolveToolCapabilities<TModel>,
168
+ > extends BaseTextAdapter<
169
+ TModel,
170
+ TProviderOptions,
171
+ TInputModalities,
172
+ GeminiMessageMetadataByModality,
173
+ TToolCapabilities
174
+ > {
175
+ override readonly kind = 'text' as const
176
+ override readonly name = 'gemini-text-interactions' as const
177
+
178
+ private readonly client: GoogleGenAI
179
+ // Tracks the most recent server-assigned interaction id per threadId
180
+ // so the adapter can chain follow-up calls on the same thread without
181
+ // the caller having to thread the id manually. Two callers rely on
182
+ // this:
183
+ // 1. The agent loop's tool-call iterations (each iteration is a new
184
+ // `chatStream` call with accumulated tool messages).
185
+ // 2. The agentic-structured composition: a `chatStream` run followed
186
+ // by `structuredOutput` on the accumulated messages.
187
+ // Cross-request chaining is the caller's job via
188
+ // `modelOptions.previous_interaction_id`. To keep stale ids from
189
+ // chaining a brand-new turn, `chatStream` evicts at the START when
190
+ // the caller signals fresh-turn intent (no caller-provided id AND a
191
+ // single user message — anything else is a follow-up). Errors evict
192
+ // immediately so a failed turn never chains into the next one.
193
+ private readonly interactionIdByThread = new Map<string, string>()
194
+
195
+ constructor(config: GeminiTextInteractionsConfig, model: TModel) {
196
+ super({}, model)
197
+ this.client = createGeminiClient(config)
198
+ }
199
+
200
+ async *chatStream(
201
+ options: TextOptions<GeminiTextInteractionsProviderOptions>,
202
+ ): AsyncIterable<StreamChunk> {
203
+ const runId = options.runId ?? generateId(this.name)
204
+ const threadId = options.threadId ?? generateId(this.name)
205
+ const timestamp = Date.now()
206
+ const { logger } = options
207
+
208
+ // Fresh-turn intent: caller didn't thread an id AND only a single
209
+ // user message is queued. Drop any stale captured id so we don't
210
+ // silently chain off a prior turn the caller doesn't know about.
211
+ // Multi-message inputs are follow-ups (agent-loop iteration or
212
+ // structuredOutput composition) and keep the Map entry.
213
+ if (
214
+ !options.modelOptions?.previous_interaction_id &&
215
+ options.messages.length === 1 &&
216
+ options.messages[0]?.role === 'user'
217
+ ) {
218
+ this.interactionIdByThread.delete(threadId)
219
+ }
220
+
221
+ // Resolve `previous_interaction_id`. Caller-provided wins; otherwise
222
+ // fall back to the id we captured during a prior iteration of this
223
+ // same agent-loop run (matched by threadId).
224
+ const effectivePreviousInteractionId =
225
+ options.modelOptions?.previous_interaction_id ??
226
+ this.interactionIdByThread.get(threadId)
227
+
228
+ let sawTerminalEvent = false
229
+ // Sentinel for the `.return()` abandonment path. Set to `true` only at
230
+ // the bottom of the `try` block — so a consumer-initiated close (via
231
+ // upstream `break` or abort) leaves it `false`, distinguishing
232
+ // abandonment from normal completion. This is the only signal that
233
+ // catches abandonment AFTER a `RUN_FINISHED(tool_calls)`, where
234
+ // `sawTerminalEvent` is `true` but the in-loop deliberately kept the
235
+ // map entry for an agent-loop iteration that will now never run.
236
+ let completedTryBlock = false
237
+ try {
238
+ const request = buildInteractionsRequest({
239
+ ...options,
240
+ modelOptions: {
241
+ ...options.modelOptions,
242
+ previous_interaction_id: effectivePreviousInteractionId,
243
+ },
244
+ })
245
+ logger.request(
246
+ `activity=chat provider=gemini-text-interactions model=${this.model} messages=${options.messages.length} tools=${options.tools?.length ?? 0} stream=true`,
247
+ {
248
+ provider: 'gemini-text-interactions',
249
+ model: this.model,
250
+ request,
251
+ },
252
+ )
253
+ const stream = (await this.client.interactions.create(
254
+ { ...request, stream: true } as GeminiInteractionsRequestBody &
255
+ Parameters<typeof this.client.interactions.create>[0],
256
+ { signal: options.abortController?.signal },
257
+ )) as AsyncIterable<InteractionSSEEvent>
258
+
259
+ for await (const chunk of translateInteractionEvents(
260
+ stream,
261
+ options.model,
262
+ runId,
263
+ threadId,
264
+ options.parentRunId,
265
+ timestamp,
266
+ this.name,
267
+ logger,
268
+ )) {
269
+ // Capture the server-assigned id so the next agent-loop
270
+ // iteration on this thread can chain off it. The CUSTOM event
271
+ // is also yielded downstream as usual — callers consume it via
272
+ // `onCustomEvent` for cross-request chaining.
273
+ //
274
+ // The yield type can't be narrowed to GeminiInteractionsStream
275
+ // here without fighting zod-passthrough variance (StreamChunk's
276
+ // CustomEvent variant carries `[k: string]: unknown`), so we
277
+ // narrow via the literal `name` and trust the typed
278
+ // construction inside `translateInteractionEvents`.
279
+ if (
280
+ chunk.type === EventType.CUSTOM &&
281
+ chunk.name === 'gemini.interactionId'
282
+ ) {
283
+ const value =
284
+ chunk.value as GeminiInteractionsCustomEventValue<'gemini.interactionId'>
285
+ this.interactionIdByThread.set(threadId, value.interactionId)
286
+ }
287
+ if (chunk.type === EventType.RUN_FINISHED) {
288
+ sawTerminalEvent = true
289
+ // Keep the captured id for follow-ups (next agent-loop
290
+ // iteration OR a structuredOutput call composing this turn).
291
+ // The next *fresh* chatStream call evicts at the top via the
292
+ // fresh-turn guard.
293
+ } else if (chunk.type === EventType.RUN_ERROR) {
294
+ sawTerminalEvent = true
295
+ this.interactionIdByThread.delete(threadId)
296
+ }
297
+ yield chunk
298
+ }
299
+
300
+ if (!sawTerminalEvent) {
301
+ // SDK stream ended without either `interaction.complete` or
302
+ // `error` — surface the truncation rather than silently leaving
303
+ // downstream consumers waiting on a `RUN_FINISHED` that will
304
+ // never come.
305
+ this.interactionIdByThread.delete(threadId)
306
+ const message =
307
+ 'Gemini Interactions stream ended without a terminal event (no interaction.complete or error)'
308
+ logger.errors('gemini-text-interactions.chatStream truncated', {
309
+ source: 'gemini-text-interactions.chatStream',
310
+ runId,
311
+ threadId,
312
+ })
313
+ yield {
314
+ type: EventType.RUN_ERROR,
315
+ runId,
316
+ model: options.model,
317
+ timestamp,
318
+ message,
319
+ error: { message },
320
+ }
321
+ }
322
+ completedTryBlock = true
323
+ } catch (error) {
324
+ this.interactionIdByThread.delete(threadId)
325
+ const message =
326
+ error instanceof Error
327
+ ? error.message
328
+ : 'An unknown error occurred during the interactions stream.'
329
+ logger.errors('gemini-text-interactions.chatStream fatal', {
330
+ error,
331
+ source: 'gemini-text-interactions.chatStream',
332
+ })
333
+ yield {
334
+ type: EventType.RUN_ERROR,
335
+ runId,
336
+ model: options.model,
337
+ timestamp,
338
+ message,
339
+ error: { message },
340
+ }
341
+ } finally {
342
+ // Abandonment cleanup — consumer `.return()` (upstream `break` /
343
+ // abort) bypasses both the truncation guard and the catch handler.
344
+ // `completedTryBlock` is the sentinel that distinguishes natural
345
+ // completion from abandonment; on abandonment we evict so a stale
346
+ // id from a half-finished turn can't chain into a follow-up. The
347
+ // catch handler also lands here with the flag false; its explicit
348
+ // delete is harmless to repeat.
349
+ if (!completedTryBlock) {
350
+ this.interactionIdByThread.delete(threadId)
351
+ }
352
+ }
353
+ }
354
+
355
+ async structuredOutput(
356
+ options: StructuredOutputOptions<GeminiTextInteractionsProviderOptions>,
357
+ ): Promise<StructuredOutputResult<unknown>> {
358
+ const { chatOptions, outputSchema } = options
359
+ const { logger } = chatOptions
360
+ const threadId = chatOptions.threadId
361
+
362
+ // Mirror the chatStream fallback: the agentic-structured flow runs
363
+ // the chat loop first and then calls structuredOutput with the
364
+ // accumulated `messages`. If any tool ran during the loop, the
365
+ // messages include assistant/tool turns; without a chained
366
+ // previous_interaction_id those would throw "cannot send prior
367
+ // conversation history on a fresh interaction".
368
+ const effectivePreviousInteractionId =
369
+ chatOptions.modelOptions?.previous_interaction_id ??
370
+ (threadId ? this.interactionIdByThread.get(threadId) : undefined)
371
+
372
+ const baseRequest = buildInteractionsRequest({
373
+ ...chatOptions,
374
+ modelOptions: {
375
+ ...chatOptions.modelOptions,
376
+ previous_interaction_id: effectivePreviousInteractionId,
377
+ },
378
+ })
379
+
380
+ const request: GeminiInteractionsRequestBody = {
381
+ ...baseRequest,
382
+ response_mime_type: 'application/json',
383
+ response_format: outputSchema,
384
+ }
385
+
386
+ try {
387
+ logger.request(
388
+ `activity=chat provider=gemini-text-interactions model=${this.model} messages=${chatOptions.messages.length} tools=${chatOptions.tools?.length ?? 0} stream=false`,
389
+ {
390
+ provider: 'gemini-text-interactions',
391
+ model: this.model,
392
+ request,
393
+ },
394
+ )
395
+ const result = (await this.client.interactions.create(
396
+ request as Parameters<typeof this.client.interactions.create>[0],
397
+ { signal: chatOptions.abortController?.signal },
398
+ )) as Interaction
399
+
400
+ const rawText = extractTextFromInteraction(result)
401
+
402
+ if (!rawText) {
403
+ throw new Error(
404
+ `Gemini Interactions returned no text output for structured-output request (status: ${result.status}). The model may have produced only tool calls or non-text content.`,
405
+ )
406
+ }
407
+
408
+ let parsed: unknown
409
+ try {
410
+ parsed = JSON.parse(rawText)
411
+ } catch {
412
+ throw new Error(
413
+ `Failed to parse structured output as JSON. Content: ${rawText.slice(0, 200)}${rawText.length > 200 ? '...' : ''}`,
414
+ )
415
+ }
416
+
417
+ return { data: parsed, rawText }
418
+ } catch (error) {
419
+ logger.errors('gemini-text-interactions.structuredOutput fatal', {
420
+ error,
421
+ source: 'gemini-text-interactions.structuredOutput',
422
+ })
423
+ // Preserve the original error as `cause` so the stack trace and any
424
+ // SDK-attached status/code/headers survive for Sentry dedup.
425
+ throw new Error(
426
+ error instanceof Error
427
+ ? error.message
428
+ : 'An unknown error occurred during structured output generation.',
429
+ { cause: error },
430
+ )
431
+ }
432
+ }
433
+ }
434
+
435
+ /** @experimental Interactions API is in Beta. */
436
+ export function createGeminiTextInteractions<TModel extends GeminiModels>(
437
+ model: TModel,
438
+ apiKey: string,
439
+ config?: Omit<GeminiTextInteractionsConfig, 'apiKey'>,
440
+ ): GeminiTextInteractionsAdapter<
441
+ TModel,
442
+ ResolveProviderOptions,
443
+ ResolveInputModalities<TModel>,
444
+ ResolveToolCapabilities<TModel>
445
+ > {
446
+ return new GeminiTextInteractionsAdapter({ apiKey, ...config }, model)
447
+ }
448
+
449
+ /** @experimental Interactions API is in Beta. */
450
+ export function geminiTextInteractions<TModel extends GeminiModels>(
451
+ model: TModel,
452
+ config?: Omit<GeminiTextInteractionsConfig, 'apiKey'>,
453
+ ): GeminiTextInteractionsAdapter<
454
+ TModel,
455
+ ResolveProviderOptions,
456
+ ResolveInputModalities<TModel>,
457
+ ResolveToolCapabilities<TModel>
458
+ > {
459
+ const apiKey = getGeminiApiKeyFromEnv()
460
+ return createGeminiTextInteractions(model, apiKey, config)
461
+ }
462
+
463
+ function buildInteractionsRequest(
464
+ options: TextOptions<GeminiTextInteractionsProviderOptions>,
465
+ ): GeminiInteractionsRequestBody {
466
+ const modelOpts = options.modelOptions
467
+
468
+ const systemInstruction =
469
+ modelOpts?.system_instruction ?? options.systemPrompts?.join('\n')
470
+
471
+ const generationConfig: Interactions.GenerationConfig = {
472
+ ...modelOpts?.generation_config,
473
+ }
474
+ if (options.temperature !== undefined) {
475
+ generationConfig.temperature = options.temperature
476
+ }
477
+ if (options.topP !== undefined) {
478
+ generationConfig.top_p = options.topP
479
+ }
480
+ if (options.maxTokens !== undefined) {
481
+ generationConfig.max_output_tokens = options.maxTokens
482
+ }
483
+
484
+ const hasGenerationConfig = Object.keys(generationConfig).length > 0
485
+
486
+ const input = convertMessagesToInteractionsInput(
487
+ options.messages,
488
+ modelOpts?.previous_interaction_id !== undefined,
489
+ )
490
+
491
+ return {
492
+ model: options.model,
493
+ input,
494
+ previous_interaction_id: modelOpts?.previous_interaction_id,
495
+ system_instruction: systemInstruction,
496
+ tools: convertToolsToInteractionsFormat(options.tools),
497
+ generation_config: hasGenerationConfig ? generationConfig : undefined,
498
+ store: modelOpts?.store,
499
+ background: modelOpts?.background,
500
+ response_modalities: modelOpts?.response_modalities,
501
+ response_format: modelOpts?.response_format,
502
+ response_mime_type: modelOpts?.response_mime_type,
503
+ }
504
+ }
505
+
506
+ // Google's Interactions API takes `input` as `Array<Step>`. Each Step
507
+ // is `{type: 'user_input' | 'function_result' | ..., ...}` — content
508
+ // blocks (text/image/etc.) live nested inside a Step's `content` array,
509
+ // they are NOT valid at the top level. Sending raw `Array<Content>`
510
+ // produces `invalid_request` / "value at top-level must be a list",
511
+ // because the API is looking for a Step list at the top level and
512
+ // gets content objects instead. The SDK's type union
513
+ // (`string | Array<Content> | Array<Turn> | ...`) is misleading; see
514
+ // https://ai.google.dev/api/interactions-api for the real Step union.
515
+ //
516
+ // When `hasPreviousInteraction` is true the server holds the transcript
517
+ // up through the last assistant turn, so we only send the steps that
518
+ // come after it (a new `user_input`, one or more `function_result`s
519
+ // continuing a tool call, etc.). Otherwise the conversation is fresh
520
+ // and only the latest user turn is supported — multi-turn replay
521
+ // without `previous_interaction_id` is not part of the API contract.
522
+ function convertMessagesToInteractionsInput(
523
+ messages: Array<ModelMessage>,
524
+ hasPreviousInteraction: boolean,
525
+ ): InteractionsRequestInput {
526
+ const toolCallIdToName = new Map<string, string>()
527
+ for (const msg of messages) {
528
+ if (msg.role === 'assistant' && msg.toolCalls) {
529
+ for (const tc of msg.toolCalls) {
530
+ toolCallIdToName.set(tc.id, tc.function.name)
531
+ }
532
+ }
533
+ }
534
+
535
+ const source = hasPreviousInteraction
536
+ ? messagesAfterLastAssistant(messages)
537
+ : messages
538
+
539
+ if (hasPreviousInteraction && source.length === 0) {
540
+ throw new Error(
541
+ 'Gemini Interactions adapter: modelOptions.previous_interaction_id was provided but no new messages were found after the last assistant turn. Append at least one user or tool message before chaining.',
542
+ )
543
+ }
544
+
545
+ if (!hasPreviousInteraction) {
546
+ const [only, ...rest] = source
547
+ if (!only) {
548
+ throw new Error('Gemini Interactions adapter: no messages to send.')
549
+ }
550
+ if (rest.length > 0) {
551
+ throw new Error(
552
+ 'Gemini Interactions adapter: cannot send prior conversation history on a fresh interaction. Either set modelOptions.previous_interaction_id to chain prior turns server-side, or trim the message list to a single new user turn. See docs/adapters/gemini.md ("Wiring with useChat") for the canonical client/server pattern.',
553
+ )
554
+ }
555
+ if (only.role !== 'user') {
556
+ throw new Error(
557
+ `Gemini Interactions adapter: the first message of a fresh interaction must be a user turn (got role="${only.role}"). Set modelOptions.previous_interaction_id to continue an existing interaction.`,
558
+ )
559
+ }
560
+ const content = messageToContentBlocks(only)
561
+ if (content.length === 0) {
562
+ throw new Error(
563
+ 'Gemini Interactions adapter: the user message produced no content blocks to send.',
564
+ )
565
+ }
566
+ return [{ type: 'user_input', content }]
567
+ }
568
+
569
+ // Chained path: each post-assistant message becomes one Step. A user
570
+ // reply maps to `user_input`; a tool reply maps to `function_result`.
571
+ // Assistant turns shouldn't appear here (sliced off above) — if one
572
+ // somehow does we skip it rather than letting it shape the wire.
573
+ const steps: Array<InteractionsStep> = []
574
+ for (const msg of source) {
575
+ if (msg.role === 'tool' && msg.toolCallId) {
576
+ const result = serializeToolResultContent(msg.content)
577
+ steps.push({
578
+ type: 'function_result',
579
+ call_id: msg.toolCallId,
580
+ name: toolCallIdToName.get(msg.toolCallId),
581
+ result,
582
+ })
583
+ } else if (msg.role === 'user') {
584
+ const content = messageToContentBlocks(msg)
585
+ if (content.length > 0) {
586
+ steps.push({ type: 'user_input', content })
587
+ }
588
+ }
589
+ }
590
+ if (steps.length === 0) {
591
+ throw new Error(
592
+ 'Gemini Interactions adapter: messages after the last assistant turn produced no steps to send.',
593
+ )
594
+ }
595
+ return steps
596
+ }
597
+
598
+ // The Interactions API's `function_result.result` field is a string. We
599
+ // fail loudly on non-string tool content rather than silently coercing
600
+ // to `''` — silent coercion meant the model lost the entire tool
601
+ // output for callers that returned content as an array (e.g. image +
602
+ // text) or `null`. If you need to send structured tool output, encode
603
+ // it yourself before passing.
604
+ function serializeToolResultContent(
605
+ content: ModelMessage['content'] | undefined,
606
+ ): string {
607
+ if (typeof content === 'string') return content
608
+ if (content === null || content === undefined) {
609
+ throw new Error(
610
+ 'Gemini Interactions adapter: tool message has no content. The Interactions API requires a string `result` on function_result steps — return a string from your tool implementation (encode JSON/multimodal output yourself).',
611
+ )
612
+ }
613
+ throw new Error(
614
+ 'Gemini Interactions adapter: tool message content must be a string (got an array of content parts). The Interactions API requires a string `result` on function_result steps — stringify multimodal tool output before returning it from your tool.',
615
+ )
616
+ }
617
+
618
+ // Extracts the content blocks (text / image / audio / video / document)
619
+ // from a single message. Tool calls and tool results live one level up
620
+ // as Steps, not as content, so they are NOT emitted here.
621
+ function messageToContentBlocks(msg: ModelMessage): Array<ContentBlock> {
622
+ const blocks: Array<ContentBlock> = []
623
+
624
+ if (Array.isArray(msg.content)) {
625
+ for (const part of msg.content) {
626
+ blocks.push(contentPartToBlock(part))
627
+ }
628
+ } else if (
629
+ typeof msg.content === 'string' &&
630
+ msg.content &&
631
+ msg.role !== 'tool'
632
+ ) {
633
+ blocks.push({ type: 'text', text: msg.content })
634
+ }
635
+
636
+ return blocks
637
+ }
638
+
639
+ function messagesAfterLastAssistant(
640
+ messages: Array<ModelMessage>,
641
+ ): Array<ModelMessage> {
642
+ for (let i = messages.length - 1; i >= 0; i--) {
643
+ if (messages[i]?.role === 'assistant') {
644
+ return messages.slice(i + 1)
645
+ }
646
+ }
647
+ return messages
648
+ }
649
+
650
+ function safeParseToolArguments(
651
+ raw: string | undefined,
652
+ logger: InternalLogger,
653
+ ): Record<string, unknown> {
654
+ if (!raw) return {}
655
+ try {
656
+ const parsed = JSON.parse(raw)
657
+ return parsed && typeof parsed === 'object' ? parsed : {}
658
+ } catch (error) {
659
+ logger.errors(
660
+ 'gemini-text-interactions.safeParseToolArguments parse failed',
661
+ {
662
+ error,
663
+ raw,
664
+ source: 'gemini-text-interactions.chatStream',
665
+ },
666
+ )
667
+ return {}
668
+ }
669
+ }
670
+
671
+ // `satisfies` pins these arrays to the SDK's narrow mime-type unions: if
672
+ // Google removes a format the build breaks, and if they add one ours keeps
673
+ // working (we just won't accept the new one until added here).
674
+ const IMAGE_MIME_TYPES = [
675
+ 'image/png',
676
+ 'image/jpeg',
677
+ 'image/webp',
678
+ 'image/heic',
679
+ 'image/heif',
680
+ ] as const satisfies ReadonlyArray<
681
+ NonNullable<Interactions.ImageContent['mime_type']>
682
+ >
683
+
684
+ const AUDIO_MIME_TYPES = [
685
+ 'audio/wav',
686
+ 'audio/mp3',
687
+ 'audio/aiff',
688
+ 'audio/aac',
689
+ 'audio/ogg',
690
+ 'audio/flac',
691
+ ] as const satisfies ReadonlyArray<
692
+ NonNullable<Interactions.AudioContent['mime_type']>
693
+ >
694
+
695
+ const VIDEO_MIME_TYPES = [
696
+ 'video/mp4',
697
+ 'video/mpeg',
698
+ 'video/mpg',
699
+ 'video/mov',
700
+ 'video/avi',
701
+ 'video/x-flv',
702
+ 'video/webm',
703
+ 'video/wmv',
704
+ 'video/3gpp',
705
+ ] as const satisfies ReadonlyArray<
706
+ NonNullable<Interactions.VideoContent['mime_type']>
707
+ >
708
+
709
+ const DOCUMENT_MIME_TYPES = [
710
+ 'application/pdf',
711
+ ] as const satisfies ReadonlyArray<
712
+ NonNullable<Interactions.DocumentContent['mime_type']>
713
+ >
714
+
715
+ function validateMime<T extends string>(
716
+ allowed: ReadonlyArray<T>,
717
+ value: string | undefined,
718
+ kind: string,
719
+ ): T | undefined {
720
+ if (value === undefined) return undefined
721
+ if ((allowed as ReadonlyArray<string>).includes(value)) {
722
+ return value as T
723
+ }
724
+ throw new Error(
725
+ `Unsupported ${kind} mime type "${value}" for the Gemini Interactions API. Allowed: ${allowed.join(', ')}.`,
726
+ )
727
+ }
728
+
729
+ function contentPartToBlock(part: ContentPart): ContentBlock {
730
+ if (part.type === 'text') {
731
+ return { type: 'text', text: part.content }
732
+ }
733
+ const isData = part.source.type === 'data'
734
+ switch (part.type) {
735
+ case 'image': {
736
+ const mime_type = validateMime(
737
+ IMAGE_MIME_TYPES,
738
+ part.source.mimeType,
739
+ 'image',
740
+ )
741
+ return isData
742
+ ? { type: 'image', data: part.source.value, mime_type }
743
+ : { type: 'image', uri: part.source.value, mime_type }
744
+ }
745
+ case 'audio': {
746
+ const mime_type = validateMime(
747
+ AUDIO_MIME_TYPES,
748
+ part.source.mimeType,
749
+ 'audio',
750
+ )
751
+ return isData
752
+ ? { type: 'audio', data: part.source.value, mime_type }
753
+ : { type: 'audio', uri: part.source.value, mime_type }
754
+ }
755
+ case 'video': {
756
+ const mime_type = validateMime(
757
+ VIDEO_MIME_TYPES,
758
+ part.source.mimeType,
759
+ 'video',
760
+ )
761
+ return isData
762
+ ? { type: 'video', data: part.source.value, mime_type }
763
+ : { type: 'video', uri: part.source.value, mime_type }
764
+ }
765
+ case 'document': {
766
+ const mime_type = validateMime(
767
+ DOCUMENT_MIME_TYPES,
768
+ part.source.mimeType,
769
+ 'document',
770
+ )
771
+ return isData
772
+ ? { type: 'document', data: part.source.value, mime_type }
773
+ : { type: 'document', uri: part.source.value, mime_type }
774
+ }
775
+ }
776
+ }
777
+
778
+ // Built-in Gemini tools use snake_case field names in the Interactions API
779
+ // that differ from the camelCase fields used on `client.models.generateContent`
780
+ // (e.g. `fileSearchStoreNames` vs `file_search_store_names`). Translate
781
+ // explicitly so callers keep using the same tool factories across adapters.
782
+ function convertToolsToInteractionsFormat<TTool extends Tool>(
783
+ tools: Array<TTool> | undefined,
784
+ ): Array<InteractionsTool> | undefined {
785
+ if (!tools || tools.length === 0) return undefined
786
+
787
+ const result: Array<InteractionsTool> = []
788
+
789
+ for (const tool of tools) {
790
+ switch (tool.name) {
791
+ case 'google_search': {
792
+ const metadata = (tool.metadata ?? {}) as {
793
+ search_types?: Array<'web_search' | 'image_search'>
794
+ }
795
+ result.push({
796
+ type: 'google_search',
797
+ ...(metadata.search_types
798
+ ? { search_types: metadata.search_types }
799
+ : {}),
800
+ })
801
+ break
802
+ }
803
+ case 'code_execution': {
804
+ result.push({ type: 'code_execution' })
805
+ break
806
+ }
807
+ case 'url_context': {
808
+ result.push({ type: 'url_context' })
809
+ break
810
+ }
811
+ case 'file_search': {
812
+ const metadata = (tool.metadata ?? {}) as {
813
+ fileSearchStoreNames?: Array<string>
814
+ topK?: number
815
+ metadataFilter?: string
816
+ }
817
+ result.push({
818
+ type: 'file_search',
819
+ ...(metadata.fileSearchStoreNames
820
+ ? { file_search_store_names: metadata.fileSearchStoreNames }
821
+ : {}),
822
+ ...(metadata.topK !== undefined ? { top_k: metadata.topK } : {}),
823
+ ...(metadata.metadataFilter !== undefined
824
+ ? { metadata_filter: metadata.metadataFilter }
825
+ : {}),
826
+ })
827
+ break
828
+ }
829
+ case 'computer_use': {
830
+ const metadata = (tool.metadata ?? {}) as {
831
+ environment?: string
832
+ excludedPredefinedFunctions?: Array<string>
833
+ }
834
+ if (metadata.environment && metadata.environment !== 'browser') {
835
+ throw new Error(
836
+ `computer_use environment "${metadata.environment}" is not supported on the Gemini Interactions API. Only "browser" is accepted.`,
837
+ )
838
+ }
839
+ result.push({
840
+ type: 'computer_use',
841
+ ...(metadata.environment
842
+ ? { environment: metadata.environment as 'browser' }
843
+ : {}),
844
+ ...(metadata.excludedPredefinedFunctions
845
+ ? {
846
+ excludedPredefinedFunctions:
847
+ metadata.excludedPredefinedFunctions,
848
+ }
849
+ : {}),
850
+ })
851
+ break
852
+ }
853
+ case 'google_search_retrieval':
854
+ throw new Error(
855
+ '`google_search_retrieval` is not supported on the Gemini Interactions API. Use `googleSearchTool()` (`google_search`) with `geminiTextInteractions()`, or call `geminiText()` for the legacy retrieval tool.',
856
+ )
857
+ case 'google_maps':
858
+ throw new Error(
859
+ '`google_maps` is not yet supported on the Gemini Interactions API. Use `geminiText()` for Google Maps grounding.',
860
+ )
861
+ case 'mcp_server':
862
+ throw new Error(
863
+ '`mcp_server` is not yet supported on the `geminiTextInteractions()` adapter.',
864
+ )
865
+ default: {
866
+ if (!tool.description) {
867
+ throw new Error(
868
+ `Tool ${tool.name} requires a description for the Gemini Interactions adapter`,
869
+ )
870
+ }
871
+ result.push({
872
+ type: 'function',
873
+ name: tool.name,
874
+ description: tool.description,
875
+ parameters: sanitizeToolParameters(
876
+ tool.inputSchema ?? { type: 'object', properties: {} },
877
+ ),
878
+ })
879
+ }
880
+ }
881
+ }
882
+
883
+ return result
884
+ }
885
+
886
+ // Map of API-level status values onto the AG-UI `finishReason` field.
887
+ // `requires_action` is the Interactions API's signal that the model
888
+ // produced one or more function calls and is waiting for results — we
889
+ // always map that to 'tool_calls' regardless of whether deltas were
890
+ // observed (a function_call may have arrived in a single delta with no
891
+ // other content). `incomplete` is the truncation signal (max_tokens
892
+ // exceeded etc.) — map to 'length'. `completed` is normal stop.
893
+ function statusToFinishReason(
894
+ status: Interaction['status'] | undefined,
895
+ sawFunctionCall: boolean,
896
+ ): 'stop' | 'length' | 'tool_calls' | null {
897
+ if (status === 'requires_action') return 'tool_calls'
898
+ if (status === 'incomplete') return 'length'
899
+ if (sawFunctionCall) return 'tool_calls'
900
+ return 'stop'
901
+ }
902
+
903
+ // Statuses that mean the interaction did not produce a usable result.
904
+ // `failed`/`cancelled` map to RUN_ERROR. `incomplete` is the model
905
+ // hitting a stop condition (max_tokens etc.) and is reported via
906
+ // `finishReason: 'length'` on RUN_FINISHED so callers can decide how to
907
+ // react without it looking like a hard error.
908
+ function statusIsError(
909
+ status: Interaction['status'] | undefined,
910
+ ): status is 'failed' | 'cancelled' {
911
+ return status === 'failed' || status === 'cancelled'
912
+ }
913
+
914
+ async function* translateInteractionEvents(
915
+ stream: AsyncIterable<InteractionSSEEvent>,
916
+ model: string,
917
+ runId: string,
918
+ threadId: string,
919
+ parentRunId: string | undefined,
920
+ timestamp: number,
921
+ adapterName: string,
922
+ logger: InternalLogger,
923
+ ): AsyncIterable<StreamChunk> {
924
+ const messageId = generateId(adapterName)
925
+ let hasEmittedRunStarted = false
926
+ let hasEmittedTextMessageStart = false
927
+ let textAccumulated = ''
928
+ let interactionId: string | undefined
929
+ let sawFunctionCall = false
930
+ const toolCalls = new Map<string, ToolCallState>()
931
+ let nextToolIndex = 0
932
+ let thinkingStepId: string | null = null
933
+ let thinkingAccumulated = ''
934
+ let reasoningMessageId: string | null = null
935
+ let hasClosedReasoning = false
936
+
937
+ const closeReasoningIfNeeded = function* (): Generator<StreamChunk> {
938
+ if (reasoningMessageId && !hasClosedReasoning) {
939
+ hasClosedReasoning = true
940
+ yield {
941
+ type: EventType.REASONING_MESSAGE_END,
942
+ messageId: reasoningMessageId,
943
+ model,
944
+ timestamp,
945
+ }
946
+ yield {
947
+ type: EventType.REASONING_END,
948
+ messageId: reasoningMessageId,
949
+ model,
950
+ timestamp,
951
+ }
952
+ // Reset so that a later `thought_summary` delta (the API
953
+ // interleaves text → thought → text on some models) opens a
954
+ // fresh reasoning block instead of re-using an already-ended
955
+ // messageId, which would violate AG-UI ordering.
956
+ thinkingStepId = null
957
+ reasoningMessageId = null
958
+ hasClosedReasoning = false
959
+ }
960
+ }
961
+
962
+ // Seals any in-flight messages and tool calls. Called both on the
963
+ // normal terminal path (`interaction.complete`) and on the error path
964
+ // (`error` SSE event + premature EOF) so the StreamProcessor never
965
+ // sees orphan TEXT_MESSAGE_START / TOOL_CALL_START / REASONING_*
966
+ // events on RUN_ERROR.
967
+ const closeOpenState = function* (): Generator<StreamChunk> {
968
+ yield* closeReasoningIfNeeded()
969
+ for (const [toolCallId, state] of toolCalls) {
970
+ if (state.ended) continue
971
+ state.ended = true
972
+ yield {
973
+ type: EventType.TOOL_CALL_END,
974
+ toolCallId,
975
+ toolName: state.name,
976
+ model,
977
+ timestamp,
978
+ input: state.args,
979
+ }
980
+ }
981
+ if (hasEmittedTextMessageStart) {
982
+ hasEmittedTextMessageStart = false
983
+ yield {
984
+ type: EventType.TEXT_MESSAGE_END,
985
+ messageId,
986
+ model,
987
+ timestamp,
988
+ }
989
+ }
990
+ }
991
+
992
+ const emitRunStartedIfNeeded = function* (): Generator<StreamChunk> {
993
+ if (!hasEmittedRunStarted) {
994
+ hasEmittedRunStarted = true
995
+ yield {
996
+ type: EventType.RUN_STARTED,
997
+ runId,
998
+ threadId,
999
+ model,
1000
+ timestamp,
1001
+ parentRunId,
1002
+ }
1003
+ }
1004
+ }
1005
+
1006
+ for await (const event of stream) {
1007
+ logger.provider(`provider=gemini-text-interactions`, { event })
1008
+ switch (event.event_type) {
1009
+ case 'interaction.start': {
1010
+ interactionId = event.interaction.id
1011
+ yield* emitRunStartedIfNeeded()
1012
+ break
1013
+ }
1014
+
1015
+ case 'content.start': {
1016
+ yield* emitRunStartedIfNeeded()
1017
+ break
1018
+ }
1019
+
1020
+ case 'content.delta': {
1021
+ yield* emitRunStartedIfNeeded()
1022
+ const delta = event.delta
1023
+ switch (delta.type) {
1024
+ case 'text': {
1025
+ yield* closeReasoningIfNeeded()
1026
+ if (!hasEmittedTextMessageStart) {
1027
+ hasEmittedTextMessageStart = true
1028
+ yield {
1029
+ type: EventType.TEXT_MESSAGE_START,
1030
+ messageId,
1031
+ model,
1032
+ timestamp,
1033
+ role: 'assistant',
1034
+ }
1035
+ }
1036
+ textAccumulated += delta.text
1037
+ yield {
1038
+ type: EventType.TEXT_MESSAGE_CONTENT,
1039
+ messageId,
1040
+ model,
1041
+ timestamp,
1042
+ delta: delta.text,
1043
+ content: textAccumulated,
1044
+ }
1045
+ break
1046
+ }
1047
+ case 'function_call': {
1048
+ yield* closeReasoningIfNeeded()
1049
+ sawFunctionCall = true
1050
+ const toolCallId = delta.id
1051
+ const deltaArgs: Record<string, unknown> =
1052
+ typeof delta.arguments === 'string'
1053
+ ? safeParseToolArguments(delta.arguments, logger)
1054
+ : delta.arguments
1055
+ let state = toolCalls.get(toolCallId)
1056
+ if (!state) {
1057
+ state = {
1058
+ name: delta.name,
1059
+ args: { ...deltaArgs },
1060
+ index: nextToolIndex++,
1061
+ started: false,
1062
+ ended: false,
1063
+ }
1064
+ toolCalls.set(toolCallId, state)
1065
+ } else {
1066
+ state.args = { ...state.args, ...deltaArgs }
1067
+ if (delta.name) state.name = delta.name
1068
+ }
1069
+ if (!state.started) {
1070
+ state.started = true
1071
+ yield {
1072
+ type: EventType.TOOL_CALL_START,
1073
+ toolCallId,
1074
+ toolCallName: state.name,
1075
+ toolName: state.name,
1076
+ model,
1077
+ timestamp,
1078
+ index: state.index,
1079
+ }
1080
+ }
1081
+ yield {
1082
+ type: EventType.TOOL_CALL_ARGS,
1083
+ toolCallId,
1084
+ model,
1085
+ timestamp,
1086
+ delta: JSON.stringify(deltaArgs),
1087
+ args: JSON.stringify(state.args),
1088
+ }
1089
+ break
1090
+ }
1091
+ case 'google_search_call': {
1092
+ yield* closeReasoningIfNeeded()
1093
+ yield {
1094
+ type: EventType.CUSTOM,
1095
+ name: 'gemini.googleSearchCall',
1096
+ value: delta,
1097
+ model,
1098
+ timestamp,
1099
+ }
1100
+ break
1101
+ }
1102
+ case 'google_search_result': {
1103
+ yield* closeReasoningIfNeeded()
1104
+ yield {
1105
+ type: EventType.CUSTOM,
1106
+ name: 'gemini.googleSearchResult',
1107
+ value: delta,
1108
+ model,
1109
+ timestamp,
1110
+ }
1111
+ break
1112
+ }
1113
+ case 'code_execution_call': {
1114
+ yield* closeReasoningIfNeeded()
1115
+ yield {
1116
+ type: EventType.CUSTOM,
1117
+ name: 'gemini.codeExecutionCall',
1118
+ value: delta,
1119
+ model,
1120
+ timestamp,
1121
+ }
1122
+ break
1123
+ }
1124
+ case 'code_execution_result': {
1125
+ yield* closeReasoningIfNeeded()
1126
+ yield {
1127
+ type: EventType.CUSTOM,
1128
+ name: 'gemini.codeExecutionResult',
1129
+ value: delta,
1130
+ model,
1131
+ timestamp,
1132
+ }
1133
+ break
1134
+ }
1135
+ case 'url_context_call': {
1136
+ yield* closeReasoningIfNeeded()
1137
+ yield {
1138
+ type: EventType.CUSTOM,
1139
+ name: 'gemini.urlContextCall',
1140
+ value: delta,
1141
+ model,
1142
+ timestamp,
1143
+ }
1144
+ break
1145
+ }
1146
+ case 'url_context_result': {
1147
+ yield* closeReasoningIfNeeded()
1148
+ yield {
1149
+ type: EventType.CUSTOM,
1150
+ name: 'gemini.urlContextResult',
1151
+ value: delta,
1152
+ model,
1153
+ timestamp,
1154
+ }
1155
+ break
1156
+ }
1157
+ case 'file_search_call': {
1158
+ yield* closeReasoningIfNeeded()
1159
+ yield {
1160
+ type: EventType.CUSTOM,
1161
+ name: 'gemini.fileSearchCall',
1162
+ value: delta,
1163
+ model,
1164
+ timestamp,
1165
+ }
1166
+ break
1167
+ }
1168
+ case 'file_search_result': {
1169
+ yield* closeReasoningIfNeeded()
1170
+ yield {
1171
+ type: EventType.CUSTOM,
1172
+ name: 'gemini.fileSearchResult',
1173
+ value: delta,
1174
+ model,
1175
+ timestamp,
1176
+ }
1177
+ break
1178
+ }
1179
+ case 'thought_summary': {
1180
+ const thoughtText =
1181
+ delta.content && 'text' in delta.content ? delta.content.text : ''
1182
+ if (!thoughtText) break
1183
+ if (thinkingStepId === null || reasoningMessageId === null) {
1184
+ thinkingStepId = generateId(adapterName)
1185
+ reasoningMessageId = generateId(adapterName)
1186
+ yield {
1187
+ type: EventType.REASONING_START,
1188
+ messageId: reasoningMessageId,
1189
+ model,
1190
+ timestamp,
1191
+ }
1192
+ yield {
1193
+ type: EventType.REASONING_MESSAGE_START,
1194
+ messageId: reasoningMessageId,
1195
+ role: 'reasoning',
1196
+ model,
1197
+ timestamp,
1198
+ }
1199
+ yield {
1200
+ type: EventType.STEP_STARTED,
1201
+ stepName: thinkingStepId,
1202
+ stepId: thinkingStepId,
1203
+ model,
1204
+ timestamp,
1205
+ stepType: 'thinking',
1206
+ }
1207
+ }
1208
+ thinkingAccumulated += thoughtText
1209
+ yield {
1210
+ type: EventType.REASONING_MESSAGE_CONTENT,
1211
+ messageId: reasoningMessageId,
1212
+ delta: thoughtText,
1213
+ model,
1214
+ timestamp,
1215
+ }
1216
+ yield {
1217
+ type: EventType.STEP_FINISHED,
1218
+ stepName: thinkingStepId,
1219
+ stepId: thinkingStepId,
1220
+ model,
1221
+ timestamp,
1222
+ delta: thoughtText,
1223
+ content: thinkingAccumulated,
1224
+ }
1225
+ break
1226
+ }
1227
+ // The following delta types are valid per the SDK type union
1228
+ // but aren't yet translated by this adapter (output modalities
1229
+ // text-only adapter shouldn't see, response-side function_result
1230
+ // / mcp_server_*, thought_signature). Falling through to the
1231
+ // observability default so SDK drift is visible.
1232
+ case 'image':
1233
+ case 'audio':
1234
+ case 'video':
1235
+ case 'document':
1236
+ case 'function_result':
1237
+ case 'mcp_server_tool_call':
1238
+ case 'mcp_server_tool_result':
1239
+ case 'thought_signature':
1240
+ default:
1241
+ logger.provider(
1242
+ `gemini-text-interactions unhandled content.delta type`,
1243
+ { delta },
1244
+ )
1245
+ break
1246
+ }
1247
+ break
1248
+ }
1249
+
1250
+ case 'content.stop':
1251
+ case 'interaction.status_update': {
1252
+ break
1253
+ }
1254
+
1255
+ case 'interaction.complete': {
1256
+ if (event.interaction.id) {
1257
+ interactionId = event.interaction.id
1258
+ }
1259
+
1260
+ yield* closeOpenState()
1261
+
1262
+ const status = event.interaction.status
1263
+ if (statusIsError(status)) {
1264
+ const message = `Gemini Interactions ${status}: the interaction ended without a usable response.`
1265
+ logger.errors(
1266
+ 'gemini-text-interactions.translateInteractionEvents non-success status',
1267
+ {
1268
+ source: 'gemini-text-interactions.chatStream',
1269
+ status,
1270
+ interactionId,
1271
+ },
1272
+ )
1273
+ yield {
1274
+ type: EventType.RUN_ERROR,
1275
+ runId,
1276
+ model,
1277
+ timestamp,
1278
+ message,
1279
+ code: status,
1280
+ error: { message, code: status },
1281
+ }
1282
+ return
1283
+ }
1284
+
1285
+ const usage = event.interaction.usage
1286
+ const finishReason = statusToFinishReason(status, sawFunctionCall)
1287
+
1288
+ if (interactionId) {
1289
+ yield {
1290
+ type: EventType.CUSTOM,
1291
+ name: 'gemini.interactionId',
1292
+ value: { interactionId },
1293
+ model,
1294
+ timestamp,
1295
+ }
1296
+ }
1297
+
1298
+ yield {
1299
+ type: EventType.RUN_FINISHED,
1300
+ runId,
1301
+ threadId,
1302
+ model,
1303
+ timestamp,
1304
+ finishReason,
1305
+ usage: usage
1306
+ ? {
1307
+ promptTokens: usage.total_input_tokens ?? 0,
1308
+ completionTokens: usage.total_output_tokens ?? 0,
1309
+ totalTokens: usage.total_tokens ?? 0,
1310
+ }
1311
+ : undefined,
1312
+ }
1313
+ return
1314
+ }
1315
+
1316
+ case 'error': {
1317
+ // Close any in-flight TEXT_MESSAGE_START / TOOL_CALL_START /
1318
+ // REASONING_* so downstream consumers don't see orphan open
1319
+ // state after RUN_ERROR.
1320
+ yield* closeOpenState()
1321
+ const rawMessage = event.error?.message
1322
+ const message =
1323
+ typeof rawMessage === 'string' && rawMessage.length > 0
1324
+ ? rawMessage
1325
+ : `Gemini Interactions error (no message): ${JSON.stringify(event.error ?? {})}`
1326
+ const rawCode = event.error?.code
1327
+ const code =
1328
+ typeof rawCode === 'string' || typeof rawCode === 'number'
1329
+ ? String(rawCode)
1330
+ : undefined
1331
+ yield {
1332
+ type: EventType.RUN_ERROR,
1333
+ runId,
1334
+ model,
1335
+ timestamp,
1336
+ message,
1337
+ code,
1338
+ error: { message, code },
1339
+ }
1340
+ return
1341
+ }
1342
+
1343
+ default:
1344
+ logger.provider(`gemini-text-interactions unhandled event_type`, {
1345
+ event,
1346
+ })
1347
+ break
1348
+ }
1349
+ }
1350
+
1351
+ // Stream ended without `interaction.complete` or `error` (both `return`
1352
+ // out of the loop). Seal any in-flight TEXT/TOOL/REASONING blocks here
1353
+ // so the truncation-fallback RUN_ERROR yielded by `chatStream` doesn't
1354
+ // leave orphan `*_START` events open downstream.
1355
+ yield* closeOpenState()
1356
+ }
1357
+
1358
+ function extractTextFromInteraction(interaction: Interaction): string {
1359
+ let text = ''
1360
+ for (const output of interaction.outputs ?? []) {
1361
+ if (output.type === 'text') {
1362
+ text += output.text
1363
+ }
1364
+ }
1365
+ return text
1366
+ }
1367
+
1368
+ // The live Interactions API rejects tool parameter schemas that contain
1369
+ // an empty `required: []` array with the misleading top-level error
1370
+ // `"value at top-level must be a list"`. Empty `properties: {}` and
1371
+ // `parameters: {}` are both fine — only the empty `required` array is
1372
+ // poison. The Zod -> JSON Schema converter (and many hand-written
1373
+ // schemas) emit `required: []` whenever a tool has zero required
1374
+ // parameters, so we strip those instances recursively before sending.
1375
+ // Non-empty `required` arrays are passed through unchanged.
1376
+ function sanitizeToolParameters(schema: unknown): unknown {
1377
+ if (!schema || typeof schema !== 'object') return schema
1378
+ if (Array.isArray(schema)) return schema.map(sanitizeToolParameters)
1379
+ const out: Record<string, unknown> = {}
1380
+ for (const [key, value] of Object.entries(schema)) {
1381
+ if (key === 'required' && Array.isArray(value) && value.length === 0) {
1382
+ continue
1383
+ }
1384
+ out[key] = sanitizeToolParameters(value)
1385
+ }
1386
+ return out
1387
+ }
1388
+
1389
+ // Re-export the stream type so consumers can import it alongside the
1390
+ // adapter from a single path: `import type { GeminiInteractionsStream }
1391
+ // from '@tanstack/ai-gemini/experimental'`.
1392
+ export type { GeminiInteractionsStream }