@tanstack/ai-gemini 0.19.1 → 0.20.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/dist/esm/adapters/audio.d.ts +1 -1
  2. package/dist/esm/adapters/audio.js.map +1 -1
  3. package/dist/esm/adapters/image.d.ts +1 -1
  4. package/dist/esm/adapters/image.js +17 -39
  5. package/dist/esm/adapters/image.js.map +1 -1
  6. package/dist/esm/adapters/summarize.d.ts +1 -1
  7. package/dist/esm/adapters/summarize.js.map +1 -1
  8. package/dist/esm/adapters/text.d.ts +1 -1
  9. package/dist/esm/adapters/text.js.map +1 -1
  10. package/dist/esm/adapters/tts.d.ts +1 -1
  11. package/dist/esm/adapters/tts.js.map +1 -1
  12. package/dist/esm/adapters/video.d.ts +60 -11
  13. package/dist/esm/adapters/video.js +205 -6
  14. package/dist/esm/adapters/video.js.map +1 -1
  15. package/dist/esm/experimental/text-interactions/adapter.d.ts +1 -1
  16. package/dist/esm/experimental/text-interactions/adapter.js.map +1 -1
  17. package/dist/esm/index.d.ts +6 -3
  18. package/dist/esm/index.js +9 -3
  19. package/dist/esm/index.js.map +1 -1
  20. package/dist/esm/model-meta.d.ts +11 -3
  21. package/dist/esm/model-meta.js +9 -1
  22. package/dist/esm/model-meta.js.map +1 -1
  23. package/dist/esm/realtime/adapter.d.ts +22 -0
  24. package/dist/esm/realtime/adapter.js +233 -0
  25. package/dist/esm/realtime/adapter.js.map +1 -0
  26. package/dist/esm/realtime/client.d.ts +98 -0
  27. package/dist/esm/realtime/client.js +389 -0
  28. package/dist/esm/realtime/client.js.map +1 -0
  29. package/dist/esm/realtime/index.d.ts +3 -0
  30. package/dist/esm/realtime/token.d.ts +26 -0
  31. package/dist/esm/realtime/token.js +39 -0
  32. package/dist/esm/realtime/token.js.map +1 -0
  33. package/dist/esm/realtime/types.d.ts +51 -0
  34. package/dist/esm/realtime/utils.d.ts +40 -0
  35. package/dist/esm/realtime/utils.js +350 -0
  36. package/dist/esm/realtime/utils.js.map +1 -0
  37. package/dist/esm/video/video-provider-options.d.ts +59 -14
  38. package/dist/esm/video/video-provider-options.js +15 -2
  39. package/dist/esm/video/video-provider-options.js.map +1 -1
  40. package/package.json +4 -4
  41. package/src/adapters/audio.ts +1 -1
  42. package/src/adapters/image.ts +25 -49
  43. package/src/adapters/summarize.ts +1 -1
  44. package/src/adapters/text.ts +1 -1
  45. package/src/adapters/tts.ts +1 -1
  46. package/src/adapters/video.ts +333 -16
  47. package/src/experimental/text-interactions/adapter.ts +2 -2
  48. package/src/index.ts +20 -2
  49. package/src/model-meta.ts +45 -2
  50. package/src/realtime/adapter.ts +311 -0
  51. package/src/realtime/client.ts +547 -0
  52. package/src/realtime/index.ts +14 -0
  53. package/src/realtime/token.ts +70 -0
  54. package/src/realtime/types.ts +94 -0
  55. package/src/realtime/utils.ts +439 -0
  56. package/src/video/video-provider-options.ts +95 -15
@@ -1,6 +1,5 @@
1
1
  import { resolveMediaPrompt } from '@tanstack/ai'
2
2
  import { BaseImageAdapter } from '@tanstack/ai/adapters'
3
- import { arrayBufferToBase64 } from '@tanstack/ai-utils'
4
3
  import {
5
4
  createGeminiClient,
6
5
  generateId,
@@ -38,7 +37,7 @@ import type {
38
37
  GoogleGenAI,
39
38
  Part,
40
39
  } from '@google/genai'
41
- import type { GeminiClientConfig } from '../utils'
40
+ import type { GeminiClientConfig } from '../utils/client'
42
41
 
43
42
  /**
44
43
  * Configuration for Gemini image adapter
@@ -199,7 +198,7 @@ export class GeminiImageAdapter<
199
198
  }),
200
199
  }
201
200
 
202
- const contents = await this.buildContents(resolved, numberOfImages)
201
+ const contents = this.buildContents(resolved, numberOfImages)
203
202
 
204
203
  const response = await this.client.models.generateContent({
205
204
  model,
@@ -220,10 +219,10 @@ export class GeminiImageAdapter<
220
219
  * The generateContent API has no numberOfImages parameter, so when more
221
220
  * than one image is requested a trailing instruction is appended.
222
221
  */
223
- private async buildContents(
222
+ private buildContents(
224
223
  resolved: ResolvedMediaPrompt,
225
224
  numberOfImages: number | undefined,
226
- ): Promise<string | Array<Content>> {
225
+ ): string | Array<Content> {
227
226
  const countInstruction =
228
227
  numberOfImages && numberOfImages > 1
229
228
  ? `Generate ${numberOfImages} distinct images.`
@@ -235,29 +234,25 @@ export class GeminiImageAdapter<
235
234
  : resolved.text
236
235
  }
237
236
 
238
- const parts: Array<Part> = await Promise.all(
239
- resolved.parts.map((part) => {
240
- if (part.type === 'text') {
241
- return Promise.resolve<Part>({ text: part.content })
242
- }
243
- if (part.type === 'image') {
244
- return this.imagePartToGeminiPart(part)
245
- }
246
- // Video / audio parts were rejected in generateImages above.
247
- throw new Error(
248
- `gemini: unsupported prompt part type "${part.type}" in image generation.`,
249
- )
250
- }),
251
- )
237
+ const parts: Array<Part> = resolved.parts.map((part) => {
238
+ if (part.type === 'text') {
239
+ return { text: part.content }
240
+ }
241
+ if (part.type === 'image') {
242
+ return this.imagePartToGeminiPart(part)
243
+ }
244
+ // Video / audio parts were rejected in generateImages above.
245
+ throw new Error(
246
+ `gemini: unsupported prompt part type "${part.type}" in image generation.`,
247
+ )
248
+ })
252
249
  if (countInstruction) {
253
250
  parts.push({ text: countInstruction })
254
251
  }
255
252
  return [{ role: 'user', parts }]
256
253
  }
257
254
 
258
- private async imagePartToGeminiPart(
259
- part: ImagePart<MediaInputMetadata>,
260
- ): Promise<Part> {
255
+ private imagePartToGeminiPart(part: ImagePart<MediaInputMetadata>): Part {
261
256
  if (part.source.type === 'data') {
262
257
  return {
263
258
  inlineData: {
@@ -266,34 +261,15 @@ export class GeminiImageAdapter<
266
261
  },
267
262
  }
268
263
  }
269
- // For URL sources, prefer passing the URL through as `fileData` when it
270
- // looks like a Google Files API URI; otherwise fetch and inline as base64.
271
- if (
272
- part.source.value.startsWith('gs://') ||
273
- /^https?:\/\/generativelanguage\.googleapis\.com\//.test(
274
- part.source.value,
275
- )
276
- ) {
277
- return {
278
- fileData: {
279
- fileUri: part.source.value,
280
- ...(part.source.mimeType && { mimeType: part.source.mimeType }),
281
- },
282
- }
283
- }
284
- const response = await fetch(part.source.value)
285
- if (!response.ok) {
286
- throw new Error(
287
- `Failed to fetch image input (${response.status} ${response.statusText}): ${part.source.value}`,
288
- )
289
- }
290
- const blob = await response.blob()
291
- const buffer = await blob.arrayBuffer()
292
- const base64 = arrayBufferToBase64(buffer)
264
+ // URL sources (public HTTPS, Files API URIs, gs://) pass through as
265
+ // `fileData` and Gemini fetches them server-side — same as the chat
266
+ // adapter. Fetching locally and inlining as base64 double-buffers the
267
+ // image and OOMs on memory-constrained runtimes (e.g. Cloudflare
268
+ // Workers).
293
269
  return {
294
- inlineData: {
295
- mimeType: part.source.mimeType || blob.type || 'image/png',
296
- data: base64,
270
+ fileData: {
271
+ fileUri: part.source.value,
272
+ mimeType: part.source.mimeType ?? 'image/jpeg',
297
273
  },
298
274
  }
299
275
  }
@@ -3,7 +3,7 @@ import { getGeminiApiKeyFromEnv } from '../utils'
3
3
  import { GeminiTextAdapter } from './text'
4
4
  import type { InferTextProviderOptions } from '@tanstack/ai/adapters'
5
5
  import type { GEMINI_MODELS } from '../model-meta'
6
- import type { GeminiClientConfig } from '../utils'
6
+ import type { GeminiClientConfig } from '../utils/client'
7
7
 
8
8
  /**
9
9
  * Configuration for Gemini summarize adapter
@@ -41,7 +41,7 @@ import type {
41
41
  GeminiMessageMetadataByModality,
42
42
  GeminiToolCallMetadata,
43
43
  } from '../message-types'
44
- import type { GeminiClientConfig } from '../utils'
44
+ import type { GeminiClientConfig } from '../utils/client'
45
45
 
46
46
  /**
47
47
  * Configuration for Gemini text adapter
@@ -9,7 +9,7 @@ import { buildGeminiUsage } from '../usage'
9
9
  import type { GEMINI_TTS_MODELS, GeminiTTSVoice } from '../model-meta'
10
10
  import type { TTSOptions, TTSResult } from '@tanstack/ai'
11
11
  import type { GoogleGenAI, SpeechConfig } from '@google/genai'
12
- import type { GeminiClientConfig } from '../utils'
12
+ import type { GeminiClientConfig } from '../utils/client'
13
13
 
14
14
  /**
15
15
  * Configuration for a single speaker in a multi-speaker dialogue.
@@ -6,13 +6,18 @@ import { resolveMediaPrompt } from '@tanstack/ai'
6
6
  import { BaseVideoAdapter, snapToDurationOption } from '@tanstack/ai/adapters'
7
7
  import { arrayBufferToBase64 } from '@tanstack/ai-utils'
8
8
  import { createGeminiClient, getGeminiApiKeyFromEnv } from '../utils'
9
- import { getGeminiVideoDurationOptions } from '../video/video-provider-options'
9
+ import {
10
+ getGeminiVideoDurationOptions,
11
+ isInteractionsVideoModel,
12
+ } from '../video/video-provider-options'
10
13
  import type { DurationOptions } from '@tanstack/ai/adapters'
11
14
  import type {
12
15
  ImagePart,
13
16
  MediaInputMetadata,
17
+ TokenUsage,
14
18
  VideoGenerationOptions,
15
19
  VideoJobResult,
20
+ VideoPart,
16
21
  VideoStatusResult,
17
22
  VideoUrlResult,
18
23
  } from '@tanstack/ai'
@@ -20,9 +25,11 @@ import type {
20
25
  GenerateVideosConfig,
21
26
  GoogleGenAI,
22
27
  Image,
28
+ Interactions,
23
29
  VideoGenerationReferenceImage,
24
30
  } from '@google/genai'
25
31
  import type {
32
+ GeminiOmniVideoProviderOptions,
26
33
  GeminiVideoModel,
27
34
  GeminiVideoModelDurationByName,
28
35
  GeminiVideoModelInputModalitiesByName,
@@ -31,14 +38,27 @@ import type {
31
38
  GeminiVideoProviderOptions,
32
39
  GeminiVideoSize,
33
40
  } from '../video/video-provider-options'
34
- import type { GeminiClientConfig } from '../utils'
41
+ import type { GeminiClientConfig } from '../utils/client'
42
+
43
+ type Interaction = Interactions.Interaction
44
+ type InteractionContent = Interactions.Content
35
45
 
36
46
  /**
37
47
  * Configuration for Gemini video adapter.
38
48
  *
39
49
  * @experimental Video generation is an experimental feature and may change.
40
50
  */
41
- export interface GeminiVideoConfig extends GeminiClientConfig {}
51
+ export interface GeminiVideoConfig extends GeminiClientConfig {
52
+ /**
53
+ * Opt into fetching HTTP(S) image URL inputs. Veo's predict API accepts
54
+ * only inline `imageBytes` or a `gcsUri`, so an HTTP(S) URL has to be
55
+ * downloaded and base64-encoded locally — which buffers the whole image in
56
+ * memory and can OOM constrained runtimes (e.g. Cloudflare Workers). When
57
+ * `false` (the default), HTTP(S) URL image inputs throw; pass a `data:` URI
58
+ * or a `gs://` reference, or set this to `true` to opt into buffering.
59
+ */
60
+ allowUrlFetch?: boolean
61
+ }
42
62
 
43
63
  /**
44
64
  * Extract a human-readable message from a long-running operation's error,
@@ -55,9 +75,17 @@ function operationErrorMessage(error: Record<string, unknown>): string {
55
75
  * Convert a TanStack image prompt part into the genai `Image` shape Veo
56
76
  * accepts: base64 `imageBytes` (data sources, data: URIs, fetched HTTP
57
77
  * URLs) or a `gcsUri` passthrough for Cloud Storage references.
78
+ *
79
+ * Unlike `generateContent` (chat / native image generation), Veo's predict
80
+ * API has no `fileData.fileUri` equivalent — `Image` only accepts
81
+ * `imageBytes` or `gcsUri`. An HTTP(S) URL therefore has to be fetched and
82
+ * inlined locally, which buffers the whole image in memory; that only happens
83
+ * when the caller opts in via `allowUrlFetch`, otherwise it throws. Prefer a
84
+ * `gs://` reference on memory-constrained runtimes.
58
85
  */
59
86
  async function imagePartToVeoImage(
60
87
  part: ImagePart<MediaInputMetadata>,
88
+ allowUrlFetch: boolean,
61
89
  ): Promise<Image> {
62
90
  if (part.source.type === 'data') {
63
91
  return {
@@ -84,6 +112,15 @@ async function imagePartToVeoImage(
84
112
  mimeType: match[1] || part.source.mimeType || 'image/png',
85
113
  }
86
114
  }
115
+ if (!allowUrlFetch) {
116
+ throw new Error(
117
+ `gemini Veo: HTTP(S) URL image inputs are not fetched by default because ` +
118
+ `Veo accepts only inline bytes, so the image would be downloaded and ` +
119
+ `buffered in memory (risking OOM on constrained runtimes). Pass a ` +
120
+ `data: URI or a gs:// reference, or set \`allowUrlFetch: true\` on the ` +
121
+ `adapter config to opt into fetching. URL: ${url}`,
122
+ )
123
+ }
87
124
  const response = await fetch(url)
88
125
  if (!response.ok) {
89
126
  throw new Error(
@@ -99,31 +136,115 @@ async function imagePartToVeoImage(
99
136
  }
100
137
 
101
138
  /**
102
- * Gemini Veo Video Generation Adapter
139
+ * Convert an image or video prompt part into an Interactions API content
140
+ * block. Data sources become inline base64 `data`; URL sources pass through
141
+ * as `uri` (Files API URIs — mirrors the Interactions text adapter).
142
+ */
143
+ function mediaPartToInteractionsContent(
144
+ part: ImagePart<MediaInputMetadata> | VideoPart<MediaInputMetadata>,
145
+ ): InteractionContent {
146
+ const mimeType = part.source.mimeType
147
+ if (part.type === 'image') {
148
+ return part.source.type === 'data'
149
+ ? { type: 'image', data: part.source.value, mime_type: mimeType }
150
+ : { type: 'image', uri: part.source.value, mime_type: mimeType }
151
+ }
152
+ return part.source.type === 'data'
153
+ ? { type: 'video', data: part.source.value, mime_type: mimeType }
154
+ : { type: 'video', uri: part.source.value, mime_type: mimeType }
155
+ }
156
+
157
+ /**
158
+ * Pull the generated video out of a completed interaction. Prefers the
159
+ * SDK's `output_video` sugar, then walks `steps` back-to-front for the last
160
+ * `model_output` step carrying a video content block (the wire shape the
161
+ * raw REST response uses).
162
+ */
163
+ function extractInteractionVideo(
164
+ interaction: Interaction,
165
+ ): { data?: string; uri?: string; mimeType: string } | undefined {
166
+ const direct = interaction.output_video
167
+ if (direct && (direct.data || direct.uri)) {
168
+ return {
169
+ data: direct.data,
170
+ uri: direct.uri,
171
+ mimeType: direct.mime_type || 'video/mp4',
172
+ }
173
+ }
174
+ const steps = interaction.steps ?? []
175
+ for (let i = steps.length - 1; i >= 0; i--) {
176
+ const step = steps[i]
177
+ if (step?.type !== 'model_output') continue
178
+ for (const block of step.content ?? []) {
179
+ if (block.type === 'video' && (block.data || block.uri)) {
180
+ return {
181
+ data: block.data,
182
+ uri: block.uri,
183
+ mimeType: block.mime_type || 'video/mp4',
184
+ }
185
+ }
186
+ }
187
+ }
188
+ return undefined
189
+ }
190
+
191
+ /**
192
+ * Map Interactions usage onto the canonical TokenUsage shape. Omni reports
193
+ * video output via `output_tokens_by_modality`; fall back to the video
194
+ * modality entry when the total is absent.
195
+ */
196
+ function interactionUsageToTokenUsage(
197
+ usage: Interaction['usage'],
198
+ ): TokenUsage | undefined {
199
+ if (!usage) return undefined
200
+ const videoTokens = usage.output_tokens_by_modality?.find(
201
+ (entry) => entry.modality === 'video',
202
+ )?.tokens
203
+ const promptTokens = usage.total_input_tokens ?? 0
204
+ const completionTokens = usage.total_output_tokens ?? videoTokens ?? 0
205
+ return {
206
+ promptTokens,
207
+ completionTokens,
208
+ totalTokens: usage.total_tokens ?? promptTokens + completionTokens,
209
+ }
210
+ }
211
+
212
+ /**
213
+ * Gemini Video Generation Adapter (Veo + Gemini Omni Flash)
103
214
  *
104
- * Tree-shakeable adapter for Google Veo video generation. Veo runs as a
105
- * long-running operation: `createVideoJob` starts the operation via the
106
- * `:predictLongRunning` endpoint, `getVideoStatus` polls it, and
107
- * `getVideoUrl` extracts the generated video's URI once it completes.
215
+ * Tree-shakeable adapter for Google video generation, routing by model:
108
216
  *
109
- * Image prompt parts are routed by `metadata.role`:
217
+ * **Veo models** run as a long-running operation: `createVideoJob` starts
218
+ * the operation via the `:predictLongRunning` endpoint, `getVideoStatus`
219
+ * polls it, and `getVideoUrl` extracts the generated video's URI once it
220
+ * completes. Image prompt parts are routed by `metadata.role`:
110
221
  * - `'start_frame'` (or the first un-roled image) → the input image the
111
222
  * video starts from
112
223
  * - `'end_frame'` → `lastFrame` (the frame the video ends on)
113
224
  * - `'reference'` / `'character'` → `referenceImages` (asset references,
114
225
  * Veo 3.1)
115
226
  *
116
- * Note: the returned video URI is served by the Gemini Files API and
227
+ * Note: the returned Veo video URI is served by the Gemini Files API and
117
228
  * requires the API key (`x-goog-api-key` header or `?key=` query
118
229
  * parameter) to download.
119
230
  *
231
+ * **Gemini Omni Flash** (`gemini-omni-flash-preview`) only serves the
232
+ * Interactions API: `createVideoJob` creates a background interaction with
233
+ * `response_modalities: ['video']`, `getVideoStatus` polls it by id, and
234
+ * `getVideoUrl` returns the inline base64 MP4 as a `data:` URL (or the
235
+ * Files API URI when the server delivers by reference). Image and video
236
+ * prompt parts are sent as interaction content blocks, grouped as images,
237
+ * then videos, then the text prompt (interleaving is not preserved); pass
238
+ * `modelOptions.previous_interaction_id` to conversationally edit a prior
239
+ * Omni generation.
240
+ *
120
241
  * @experimental Video generation is an experimental feature and may change.
121
242
  */
122
243
  export class GeminiVideoAdapter<
123
244
  TModel extends GeminiVideoModel,
124
245
  > extends BaseVideoAdapter<
125
246
  TModel,
126
- GeminiVideoProviderOptions,
247
+ GeminiVideoModelProviderOptionsByName[TModel],
127
248
  GeminiVideoModelProviderOptionsByName,
128
249
  GeminiVideoModelSizeByName,
129
250
  GeminiVideoModelInputModalitiesByName,
@@ -132,26 +253,35 @@ export class GeminiVideoAdapter<
132
253
  readonly name = 'gemini' as const
133
254
 
134
255
  protected client: GoogleGenAI
256
+ private readonly allowUrlFetch: boolean
135
257
 
136
258
  constructor(config: GeminiVideoConfig, model: TModel) {
137
259
  super({}, model)
138
260
  this.client = createGeminiClient(config)
261
+ this.allowUrlFetch = config.allowUrlFetch ?? false
139
262
  }
140
263
 
141
264
  async createVideoJob(
142
265
  options: VideoGenerationOptions<
143
- GeminiVideoProviderOptions,
266
+ GeminiVideoModelProviderOptionsByName[TModel],
144
267
  GeminiVideoSize,
145
268
  GeminiVideoModelDurationByName[TModel]
146
269
  >,
147
270
  ): Promise<VideoJobResult> {
148
- const { prompt, size, duration, modelOptions, logger } = options
271
+ const { prompt, size, duration, logger } = options
149
272
 
150
273
  logger.request(
151
274
  `activity=video.create provider=${this.name} model=${this.model} size=${size ?? 'default'} duration=${duration ?? 'default'}`,
152
275
  { provider: this.name, model: this.model },
153
276
  )
154
277
 
278
+ if (isInteractionsVideoModel(this.model)) {
279
+ return await this.createInteractionsVideoJob(options)
280
+ }
281
+ const modelOptions = options.modelOptions as
282
+ | GeminiVideoProviderOptions
283
+ | undefined
284
+
155
285
  try {
156
286
  const resolved = resolveMediaPrompt(prompt)
157
287
 
@@ -201,6 +331,99 @@ export class GeminiVideoAdapter<
201
331
  }
202
332
  }
203
333
 
334
+ /**
335
+ * Gemini Omni Flash job creation via the Interactions API. Creates a
336
+ * background interaction requesting video output; the interaction id is
337
+ * the job id polled by `getVideoStatus` / `getVideoUrl`.
338
+ */
339
+ private async createInteractionsVideoJob(
340
+ options: VideoGenerationOptions<
341
+ GeminiVideoModelProviderOptionsByName[TModel],
342
+ GeminiVideoSize,
343
+ GeminiVideoModelDurationByName[TModel]
344
+ >,
345
+ ): Promise<VideoJobResult> {
346
+ const { prompt, size, duration, logger } = options
347
+ const modelOptions = options.modelOptions as
348
+ | GeminiOmniVideoProviderOptions
349
+ | undefined
350
+
351
+ try {
352
+ const resolved = resolveMediaPrompt(prompt)
353
+
354
+ if (resolved.audios.length > 0) {
355
+ throw new Error(
356
+ `${this.name}.createVideoJob does not support audio prompt parts (model: ${this.model}).`,
357
+ )
358
+ }
359
+
360
+ const content: Array<InteractionContent> = [
361
+ ...resolved.images.map(mediaPartToInteractionsContent),
362
+ ...resolved.videos.map(mediaPartToInteractionsContent),
363
+ ]
364
+ if (resolved.text) {
365
+ content.push({ type: 'text', text: resolved.text })
366
+ }
367
+ if (content.length === 0) {
368
+ throw new Error(
369
+ `${this.name}.createVideoJob: the prompt produced no content to send (model: ${this.model}).`,
370
+ )
371
+ }
372
+
373
+ // Reject out-of-range durations locally rather than snapping (which
374
+ // would silently change the clip length the caller asked for) or
375
+ // letting the live API reject them after the round trip.
376
+ const durations = this.availableDurations()
377
+ if (
378
+ duration !== undefined &&
379
+ durations.kind === 'range' &&
380
+ (duration < durations.min || duration > durations.max)
381
+ ) {
382
+ throw new Error(
383
+ `${this.name}.createVideoJob: duration ${duration}s is outside the ${durations.min}–${durations.max}s range supported by ${this.model}. Use snapDuration() to snap arbitrary values into range.`,
384
+ )
385
+ }
386
+
387
+ // Aspect ratio and clip length ride on `response_format`. Duration is
388
+ // a `"<seconds>s"` string, accepted anywhere in the 3–10s range
389
+ // (fractional included) and defaulting to 10s when omitted — verified
390
+ // against the live API; the docs don't publish the range constraints.
391
+ const responseFormat =
392
+ size !== undefined || duration !== undefined
393
+ ? {
394
+ response_format: {
395
+ type: 'video' as const,
396
+ ...(size !== undefined && { aspect_ratio: size }),
397
+ ...(duration !== undefined && { duration: `${duration}s` }),
398
+ },
399
+ }
400
+ : {}
401
+
402
+ const interaction = await this.client.interactions.create({
403
+ ...modelOptions,
404
+ model: this.model,
405
+ input: [{ type: 'user_input', content }],
406
+ response_modalities: ['video'],
407
+ background: true,
408
+ ...responseFormat,
409
+ })
410
+
411
+ if (!interaction.id) {
412
+ throw new Error(
413
+ 'Gemini Omni did not return an interaction id for the video generation job.',
414
+ )
415
+ }
416
+
417
+ return { jobId: interaction.id, model: this.model }
418
+ } catch (error) {
419
+ logger.errors(`${this.name}.createVideoJob fatal`, {
420
+ error,
421
+ source: `${this.name}.createVideoJob`,
422
+ })
423
+ throw error
424
+ }
425
+ }
426
+
204
427
  /**
205
428
  * Route image prompt parts onto Veo's request fields by `metadata.role`.
206
429
  */
@@ -224,13 +447,13 @@ export class GeminiVideoAdapter<
224
447
  `${this.name}: Veo accepts at most one 'end_frame' image.`,
225
448
  )
226
449
  }
227
- lastFrame = await imagePartToVeoImage(part)
450
+ lastFrame = await imagePartToVeoImage(part, this.allowUrlFetch)
228
451
  break
229
452
  }
230
453
  case 'reference':
231
454
  case 'character': {
232
455
  referenceImages.push({
233
- image: await imagePartToVeoImage(part),
456
+ image: await imagePartToVeoImage(part, this.allowUrlFetch),
234
457
  referenceType: VideoGenerationReferenceType.ASSET,
235
458
  })
236
459
  break
@@ -242,7 +465,7 @@ export class GeminiVideoAdapter<
242
465
  `${this.name}: Veo accepts at most one starting image; received multiple 'start_frame'/un-roled images. Use metadata.role ('end_frame', 'reference') to disambiguate the others.`,
243
466
  )
244
467
  }
245
- image = await imagePartToVeoImage(part)
468
+ image = await imagePartToVeoImage(part, this.allowUrlFetch)
246
469
  break
247
470
  }
248
471
  case 'mask':
@@ -257,6 +480,9 @@ export class GeminiVideoAdapter<
257
480
  }
258
481
 
259
482
  async getVideoStatus(jobId: string): Promise<VideoStatusResult> {
483
+ if (isInteractionsVideoModel(this.model)) {
484
+ return await this.getInteractionsVideoStatus(jobId)
485
+ }
260
486
  const operation = await this.getOperation(jobId)
261
487
 
262
488
  if (!operation.done) {
@@ -289,7 +515,55 @@ export class GeminiVideoAdapter<
289
515
  return { jobId, status: 'completed' }
290
516
  }
291
517
 
518
+ /**
519
+ * Poll an Omni background interaction. `in_progress` maps to
520
+ * 'processing'; a `completed` interaction with no video content (e.g.
521
+ * filtered output) is surfaced as a failure so `getVideoUrl` doesn't
522
+ * throw on an empty response. `requires_action` also fails: the adapter
523
+ * never sends tools, so it can only arise via
524
+ * `previous_interaction_id` chaining onto a tool-bearing interaction —
525
+ * and such an interaction never progresses without a client response,
526
+ * so polling it would spin until timeout.
527
+ */
528
+ private async getInteractionsVideoStatus(
529
+ jobId: string,
530
+ ): Promise<VideoStatusResult> {
531
+ const interaction = await this.getInteraction(jobId)
532
+ const status = interaction.status
533
+
534
+ if (status === 'in_progress') {
535
+ return { jobId, status: 'processing' }
536
+ }
537
+ if (status === 'requires_action') {
538
+ return {
539
+ jobId,
540
+ status: 'failed',
541
+ error:
542
+ 'Gemini Omni interaction is waiting on a client action (tool response), which the video jobs flow does not support.',
543
+ }
544
+ }
545
+ if (status === 'completed') {
546
+ if (!extractInteractionVideo(interaction)) {
547
+ return {
548
+ jobId,
549
+ status: 'failed',
550
+ error:
551
+ 'Gemini Omni completed the interaction without returning a video (the output may have been filtered).',
552
+ }
553
+ }
554
+ return { jobId, status: 'completed' }
555
+ }
556
+ return {
557
+ jobId,
558
+ status: 'failed',
559
+ error: `Gemini Omni video generation ended with status "${status}".`,
560
+ }
561
+ }
562
+
292
563
  async getVideoUrl(jobId: string): Promise<VideoUrlResult> {
564
+ if (isInteractionsVideoModel(this.model)) {
565
+ return await this.getInteractionsVideoUrl(jobId)
566
+ }
293
567
  const operation = await this.getOperation(jobId)
294
568
 
295
569
  if (!operation.done) {
@@ -317,6 +591,42 @@ export class GeminiVideoAdapter<
317
591
  return { jobId, url: uri }
318
592
  }
319
593
 
594
+ /**
595
+ * Extract the finished Omni video. Inline base64 output (the API default)
596
+ * becomes a `data:` URL — matching the OpenAI Sora adapter's inline
597
+ * delivery — and URI delivery passes through (Files API URIs need the API
598
+ * key to download, like Veo). Usage carries the video-modality output
599
+ * tokens (Omni bills per second of video, reported as tokens).
600
+ */
601
+ private async getInteractionsVideoUrl(
602
+ jobId: string,
603
+ ): Promise<VideoUrlResult> {
604
+ const interaction = await this.getInteraction(jobId)
605
+ const status = interaction.status
606
+
607
+ if (status === 'in_progress') {
608
+ throw new Error(
609
+ `Video is not ready yet. Check status first. Job ID: ${jobId}`,
610
+ )
611
+ }
612
+ if (status !== 'completed') {
613
+ throw new Error(
614
+ `Video generation failed: Gemini Omni interaction ended with status "${status}". Job ID: ${jobId}`,
615
+ )
616
+ }
617
+
618
+ const video = extractInteractionVideo(interaction)
619
+ if (!video) {
620
+ throw new Error(
621
+ `Video not found in interaction response (the output may have been filtered). Job ID: ${jobId}`,
622
+ )
623
+ }
624
+
625
+ const usage = interactionUsageToTokenUsage(interaction.usage)
626
+ const url = video.uri ?? `data:${video.mimeType};base64,${video.data}`
627
+ return { jobId, url, ...(usage && { usage }) }
628
+ }
629
+
320
630
  override availableDurations(): DurationOptions<
321
631
  GeminiVideoModelDurationByName[TModel]
322
632
  > {
@@ -340,6 +650,13 @@ export class GeminiVideoAdapter<
340
650
  operation.name = jobId
341
651
  return await this.client.operations.getVideosOperation({ operation })
342
652
  }
653
+
654
+ /**
655
+ * Fetch an Omni background interaction by id.
656
+ */
657
+ private async getInteraction(jobId: string): Promise<Interaction> {
658
+ return await this.client.interactions.get(jobId)
659
+ }
343
660
  }
344
661
 
345
662
  /**
@@ -5,7 +5,7 @@ import {
5
5
  createGeminiClient,
6
6
  generateId,
7
7
  getGeminiApiKeyFromEnv,
8
- } from '../../utils'
8
+ } from '../../utils/client'
9
9
  import type { InternalLogger } from '@tanstack/ai/adapter-internals'
10
10
  import type {
11
11
  GeminiChatModelToolCapabilitiesByName,
@@ -33,7 +33,7 @@ import type {
33
33
  } from './events'
34
34
  import type { ExternalTextInteractionsProviderOptions } from './provider-options'
35
35
  import type { GeminiMessageMetadataByModality } from '../../message-types'
36
- import type { GeminiClientConfig } from '../../utils'
36
+ import type { GeminiClientConfig } from '../../utils/client'
37
37
 
38
38
  type Interaction = Interactions.Interaction
39
39
  type InteractionSSEEvent = Interactions.InteractionSSEEvent