@tanstack/ai-gemini 0.19.0 → 0.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/audio.d.ts +1 -1
- package/dist/esm/adapters/audio.js.map +1 -1
- package/dist/esm/adapters/image.d.ts +1 -1
- package/dist/esm/adapters/image.js +17 -39
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/adapters/summarize.d.ts +1 -1
- package/dist/esm/adapters/summarize.js.map +1 -1
- package/dist/esm/adapters/text.d.ts +1 -1
- package/dist/esm/adapters/text.js.map +1 -1
- package/dist/esm/adapters/tts.d.ts +1 -1
- package/dist/esm/adapters/tts.js.map +1 -1
- package/dist/esm/adapters/video.d.ts +60 -11
- package/dist/esm/adapters/video.js +205 -6
- package/dist/esm/adapters/video.js.map +1 -1
- package/dist/esm/experimental/text-interactions/adapter.d.ts +1 -1
- package/dist/esm/experimental/text-interactions/adapter.js.map +1 -1
- package/dist/esm/index.d.ts +6 -3
- package/dist/esm/index.js +9 -3
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/model-meta.d.ts +11 -3
- package/dist/esm/model-meta.js +9 -1
- package/dist/esm/model-meta.js.map +1 -1
- package/dist/esm/realtime/adapter.d.ts +22 -0
- package/dist/esm/realtime/adapter.js +233 -0
- package/dist/esm/realtime/adapter.js.map +1 -0
- package/dist/esm/realtime/client.d.ts +98 -0
- package/dist/esm/realtime/client.js +389 -0
- package/dist/esm/realtime/client.js.map +1 -0
- package/dist/esm/realtime/index.d.ts +3 -0
- package/dist/esm/realtime/token.d.ts +26 -0
- package/dist/esm/realtime/token.js +39 -0
- package/dist/esm/realtime/token.js.map +1 -0
- package/dist/esm/realtime/types.d.ts +51 -0
- package/dist/esm/realtime/utils.d.ts +40 -0
- package/dist/esm/realtime/utils.js +350 -0
- package/dist/esm/realtime/utils.js.map +1 -0
- package/dist/esm/video/video-provider-options.d.ts +59 -14
- package/dist/esm/video/video-provider-options.js +15 -2
- package/dist/esm/video/video-provider-options.js.map +1 -1
- package/package.json +4 -4
- package/src/adapters/audio.ts +1 -1
- package/src/adapters/image.ts +25 -49
- package/src/adapters/summarize.ts +1 -1
- package/src/adapters/text.ts +1 -1
- package/src/adapters/tts.ts +1 -1
- package/src/adapters/video.ts +333 -16
- package/src/experimental/text-interactions/adapter.ts +2 -2
- package/src/index.ts +20 -2
- package/src/model-meta.ts +45 -2
- package/src/realtime/adapter.ts +311 -0
- package/src/realtime/client.ts +547 -0
- package/src/realtime/index.ts +14 -0
- package/src/realtime/token.ts +70 -0
- package/src/realtime/types.ts +94 -0
- package/src/realtime/utils.ts +439 -0
- package/src/video/video-provider-options.ts +95 -15
package/src/adapters/image.ts
CHANGED
|
@@ -1,6 +1,5 @@
|
|
|
1
1
|
import { resolveMediaPrompt } from '@tanstack/ai'
|
|
2
2
|
import { BaseImageAdapter } from '@tanstack/ai/adapters'
|
|
3
|
-
import { arrayBufferToBase64 } from '@tanstack/ai-utils'
|
|
4
3
|
import {
|
|
5
4
|
createGeminiClient,
|
|
6
5
|
generateId,
|
|
@@ -38,7 +37,7 @@ import type {
|
|
|
38
37
|
GoogleGenAI,
|
|
39
38
|
Part,
|
|
40
39
|
} from '@google/genai'
|
|
41
|
-
import type { GeminiClientConfig } from '../utils'
|
|
40
|
+
import type { GeminiClientConfig } from '../utils/client'
|
|
42
41
|
|
|
43
42
|
/**
|
|
44
43
|
* Configuration for Gemini image adapter
|
|
@@ -199,7 +198,7 @@ export class GeminiImageAdapter<
|
|
|
199
198
|
}),
|
|
200
199
|
}
|
|
201
200
|
|
|
202
|
-
const contents =
|
|
201
|
+
const contents = this.buildContents(resolved, numberOfImages)
|
|
203
202
|
|
|
204
203
|
const response = await this.client.models.generateContent({
|
|
205
204
|
model,
|
|
@@ -220,10 +219,10 @@ export class GeminiImageAdapter<
|
|
|
220
219
|
* The generateContent API has no numberOfImages parameter, so when more
|
|
221
220
|
* than one image is requested a trailing instruction is appended.
|
|
222
221
|
*/
|
|
223
|
-
private
|
|
222
|
+
private buildContents(
|
|
224
223
|
resolved: ResolvedMediaPrompt,
|
|
225
224
|
numberOfImages: number | undefined,
|
|
226
|
-
):
|
|
225
|
+
): string | Array<Content> {
|
|
227
226
|
const countInstruction =
|
|
228
227
|
numberOfImages && numberOfImages > 1
|
|
229
228
|
? `Generate ${numberOfImages} distinct images.`
|
|
@@ -235,29 +234,25 @@ export class GeminiImageAdapter<
|
|
|
235
234
|
: resolved.text
|
|
236
235
|
}
|
|
237
236
|
|
|
238
|
-
const parts: Array<Part> =
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
}),
|
|
251
|
-
)
|
|
237
|
+
const parts: Array<Part> = resolved.parts.map((part) => {
|
|
238
|
+
if (part.type === 'text') {
|
|
239
|
+
return { text: part.content }
|
|
240
|
+
}
|
|
241
|
+
if (part.type === 'image') {
|
|
242
|
+
return this.imagePartToGeminiPart(part)
|
|
243
|
+
}
|
|
244
|
+
// Video / audio parts were rejected in generateImages above.
|
|
245
|
+
throw new Error(
|
|
246
|
+
`gemini: unsupported prompt part type "${part.type}" in image generation.`,
|
|
247
|
+
)
|
|
248
|
+
})
|
|
252
249
|
if (countInstruction) {
|
|
253
250
|
parts.push({ text: countInstruction })
|
|
254
251
|
}
|
|
255
252
|
return [{ role: 'user', parts }]
|
|
256
253
|
}
|
|
257
254
|
|
|
258
|
-
private
|
|
259
|
-
part: ImagePart<MediaInputMetadata>,
|
|
260
|
-
): Promise<Part> {
|
|
255
|
+
private imagePartToGeminiPart(part: ImagePart<MediaInputMetadata>): Part {
|
|
261
256
|
if (part.source.type === 'data') {
|
|
262
257
|
return {
|
|
263
258
|
inlineData: {
|
|
@@ -266,34 +261,15 @@ export class GeminiImageAdapter<
|
|
|
266
261
|
},
|
|
267
262
|
}
|
|
268
263
|
}
|
|
269
|
-
//
|
|
270
|
-
//
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
part.source.value,
|
|
275
|
-
)
|
|
276
|
-
) {
|
|
277
|
-
return {
|
|
278
|
-
fileData: {
|
|
279
|
-
fileUri: part.source.value,
|
|
280
|
-
...(part.source.mimeType && { mimeType: part.source.mimeType }),
|
|
281
|
-
},
|
|
282
|
-
}
|
|
283
|
-
}
|
|
284
|
-
const response = await fetch(part.source.value)
|
|
285
|
-
if (!response.ok) {
|
|
286
|
-
throw new Error(
|
|
287
|
-
`Failed to fetch image input (${response.status} ${response.statusText}): ${part.source.value}`,
|
|
288
|
-
)
|
|
289
|
-
}
|
|
290
|
-
const blob = await response.blob()
|
|
291
|
-
const buffer = await blob.arrayBuffer()
|
|
292
|
-
const base64 = arrayBufferToBase64(buffer)
|
|
264
|
+
// URL sources (public HTTPS, Files API URIs, gs://) pass through as
|
|
265
|
+
// `fileData` and Gemini fetches them server-side — same as the chat
|
|
266
|
+
// adapter. Fetching locally and inlining as base64 double-buffers the
|
|
267
|
+
// image and OOMs on memory-constrained runtimes (e.g. Cloudflare
|
|
268
|
+
// Workers).
|
|
293
269
|
return {
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
270
|
+
fileData: {
|
|
271
|
+
fileUri: part.source.value,
|
|
272
|
+
mimeType: part.source.mimeType ?? 'image/jpeg',
|
|
297
273
|
},
|
|
298
274
|
}
|
|
299
275
|
}
|
|
@@ -3,7 +3,7 @@ import { getGeminiApiKeyFromEnv } from '../utils'
|
|
|
3
3
|
import { GeminiTextAdapter } from './text'
|
|
4
4
|
import type { InferTextProviderOptions } from '@tanstack/ai/adapters'
|
|
5
5
|
import type { GEMINI_MODELS } from '../model-meta'
|
|
6
|
-
import type { GeminiClientConfig } from '../utils'
|
|
6
|
+
import type { GeminiClientConfig } from '../utils/client'
|
|
7
7
|
|
|
8
8
|
/**
|
|
9
9
|
* Configuration for Gemini summarize adapter
|
package/src/adapters/text.ts
CHANGED
|
@@ -41,7 +41,7 @@ import type {
|
|
|
41
41
|
GeminiMessageMetadataByModality,
|
|
42
42
|
GeminiToolCallMetadata,
|
|
43
43
|
} from '../message-types'
|
|
44
|
-
import type { GeminiClientConfig } from '../utils'
|
|
44
|
+
import type { GeminiClientConfig } from '../utils/client'
|
|
45
45
|
|
|
46
46
|
/**
|
|
47
47
|
* Configuration for Gemini text adapter
|
package/src/adapters/tts.ts
CHANGED
|
@@ -9,7 +9,7 @@ import { buildGeminiUsage } from '../usage'
|
|
|
9
9
|
import type { GEMINI_TTS_MODELS, GeminiTTSVoice } from '../model-meta'
|
|
10
10
|
import type { TTSOptions, TTSResult } from '@tanstack/ai'
|
|
11
11
|
import type { GoogleGenAI, SpeechConfig } from '@google/genai'
|
|
12
|
-
import type { GeminiClientConfig } from '../utils'
|
|
12
|
+
import type { GeminiClientConfig } from '../utils/client'
|
|
13
13
|
|
|
14
14
|
/**
|
|
15
15
|
* Configuration for a single speaker in a multi-speaker dialogue.
|
package/src/adapters/video.ts
CHANGED
|
@@ -6,13 +6,18 @@ import { resolveMediaPrompt } from '@tanstack/ai'
|
|
|
6
6
|
import { BaseVideoAdapter, snapToDurationOption } from '@tanstack/ai/adapters'
|
|
7
7
|
import { arrayBufferToBase64 } from '@tanstack/ai-utils'
|
|
8
8
|
import { createGeminiClient, getGeminiApiKeyFromEnv } from '../utils'
|
|
9
|
-
import {
|
|
9
|
+
import {
|
|
10
|
+
getGeminiVideoDurationOptions,
|
|
11
|
+
isInteractionsVideoModel,
|
|
12
|
+
} from '../video/video-provider-options'
|
|
10
13
|
import type { DurationOptions } from '@tanstack/ai/adapters'
|
|
11
14
|
import type {
|
|
12
15
|
ImagePart,
|
|
13
16
|
MediaInputMetadata,
|
|
17
|
+
TokenUsage,
|
|
14
18
|
VideoGenerationOptions,
|
|
15
19
|
VideoJobResult,
|
|
20
|
+
VideoPart,
|
|
16
21
|
VideoStatusResult,
|
|
17
22
|
VideoUrlResult,
|
|
18
23
|
} from '@tanstack/ai'
|
|
@@ -20,9 +25,11 @@ import type {
|
|
|
20
25
|
GenerateVideosConfig,
|
|
21
26
|
GoogleGenAI,
|
|
22
27
|
Image,
|
|
28
|
+
Interactions,
|
|
23
29
|
VideoGenerationReferenceImage,
|
|
24
30
|
} from '@google/genai'
|
|
25
31
|
import type {
|
|
32
|
+
GeminiOmniVideoProviderOptions,
|
|
26
33
|
GeminiVideoModel,
|
|
27
34
|
GeminiVideoModelDurationByName,
|
|
28
35
|
GeminiVideoModelInputModalitiesByName,
|
|
@@ -31,14 +38,27 @@ import type {
|
|
|
31
38
|
GeminiVideoProviderOptions,
|
|
32
39
|
GeminiVideoSize,
|
|
33
40
|
} from '../video/video-provider-options'
|
|
34
|
-
import type { GeminiClientConfig } from '../utils'
|
|
41
|
+
import type { GeminiClientConfig } from '../utils/client'
|
|
42
|
+
|
|
43
|
+
type Interaction = Interactions.Interaction
|
|
44
|
+
type InteractionContent = Interactions.Content
|
|
35
45
|
|
|
36
46
|
/**
|
|
37
47
|
* Configuration for Gemini video adapter.
|
|
38
48
|
*
|
|
39
49
|
* @experimental Video generation is an experimental feature and may change.
|
|
40
50
|
*/
|
|
41
|
-
export interface GeminiVideoConfig extends GeminiClientConfig {
|
|
51
|
+
export interface GeminiVideoConfig extends GeminiClientConfig {
|
|
52
|
+
/**
|
|
53
|
+
* Opt into fetching HTTP(S) image URL inputs. Veo's predict API accepts
|
|
54
|
+
* only inline `imageBytes` or a `gcsUri`, so an HTTP(S) URL has to be
|
|
55
|
+
* downloaded and base64-encoded locally — which buffers the whole image in
|
|
56
|
+
* memory and can OOM constrained runtimes (e.g. Cloudflare Workers). When
|
|
57
|
+
* `false` (the default), HTTP(S) URL image inputs throw; pass a `data:` URI
|
|
58
|
+
* or a `gs://` reference, or set this to `true` to opt into buffering.
|
|
59
|
+
*/
|
|
60
|
+
allowUrlFetch?: boolean
|
|
61
|
+
}
|
|
42
62
|
|
|
43
63
|
/**
|
|
44
64
|
* Extract a human-readable message from a long-running operation's error,
|
|
@@ -55,9 +75,17 @@ function operationErrorMessage(error: Record<string, unknown>): string {
|
|
|
55
75
|
* Convert a TanStack image prompt part into the genai `Image` shape Veo
|
|
56
76
|
* accepts: base64 `imageBytes` (data sources, data: URIs, fetched HTTP
|
|
57
77
|
* URLs) or a `gcsUri` passthrough for Cloud Storage references.
|
|
78
|
+
*
|
|
79
|
+
* Unlike `generateContent` (chat / native image generation), Veo's predict
|
|
80
|
+
* API has no `fileData.fileUri` equivalent — `Image` only accepts
|
|
81
|
+
* `imageBytes` or `gcsUri`. An HTTP(S) URL therefore has to be fetched and
|
|
82
|
+
* inlined locally, which buffers the whole image in memory; that only happens
|
|
83
|
+
* when the caller opts in via `allowUrlFetch`, otherwise it throws. Prefer a
|
|
84
|
+
* `gs://` reference on memory-constrained runtimes.
|
|
58
85
|
*/
|
|
59
86
|
async function imagePartToVeoImage(
|
|
60
87
|
part: ImagePart<MediaInputMetadata>,
|
|
88
|
+
allowUrlFetch: boolean,
|
|
61
89
|
): Promise<Image> {
|
|
62
90
|
if (part.source.type === 'data') {
|
|
63
91
|
return {
|
|
@@ -84,6 +112,15 @@ async function imagePartToVeoImage(
|
|
|
84
112
|
mimeType: match[1] || part.source.mimeType || 'image/png',
|
|
85
113
|
}
|
|
86
114
|
}
|
|
115
|
+
if (!allowUrlFetch) {
|
|
116
|
+
throw new Error(
|
|
117
|
+
`gemini Veo: HTTP(S) URL image inputs are not fetched by default because ` +
|
|
118
|
+
`Veo accepts only inline bytes, so the image would be downloaded and ` +
|
|
119
|
+
`buffered in memory (risking OOM on constrained runtimes). Pass a ` +
|
|
120
|
+
`data: URI or a gs:// reference, or set \`allowUrlFetch: true\` on the ` +
|
|
121
|
+
`adapter config to opt into fetching. URL: ${url}`,
|
|
122
|
+
)
|
|
123
|
+
}
|
|
87
124
|
const response = await fetch(url)
|
|
88
125
|
if (!response.ok) {
|
|
89
126
|
throw new Error(
|
|
@@ -99,31 +136,115 @@ async function imagePartToVeoImage(
|
|
|
99
136
|
}
|
|
100
137
|
|
|
101
138
|
/**
|
|
102
|
-
*
|
|
139
|
+
* Convert an image or video prompt part into an Interactions API content
|
|
140
|
+
* block. Data sources become inline base64 `data`; URL sources pass through
|
|
141
|
+
* as `uri` (Files API URIs — mirrors the Interactions text adapter).
|
|
142
|
+
*/
|
|
143
|
+
function mediaPartToInteractionsContent(
|
|
144
|
+
part: ImagePart<MediaInputMetadata> | VideoPart<MediaInputMetadata>,
|
|
145
|
+
): InteractionContent {
|
|
146
|
+
const mimeType = part.source.mimeType
|
|
147
|
+
if (part.type === 'image') {
|
|
148
|
+
return part.source.type === 'data'
|
|
149
|
+
? { type: 'image', data: part.source.value, mime_type: mimeType }
|
|
150
|
+
: { type: 'image', uri: part.source.value, mime_type: mimeType }
|
|
151
|
+
}
|
|
152
|
+
return part.source.type === 'data'
|
|
153
|
+
? { type: 'video', data: part.source.value, mime_type: mimeType }
|
|
154
|
+
: { type: 'video', uri: part.source.value, mime_type: mimeType }
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* Pull the generated video out of a completed interaction. Prefers the
|
|
159
|
+
* SDK's `output_video` sugar, then walks `steps` back-to-front for the last
|
|
160
|
+
* `model_output` step carrying a video content block (the wire shape the
|
|
161
|
+
* raw REST response uses).
|
|
162
|
+
*/
|
|
163
|
+
function extractInteractionVideo(
|
|
164
|
+
interaction: Interaction,
|
|
165
|
+
): { data?: string; uri?: string; mimeType: string } | undefined {
|
|
166
|
+
const direct = interaction.output_video
|
|
167
|
+
if (direct && (direct.data || direct.uri)) {
|
|
168
|
+
return {
|
|
169
|
+
data: direct.data,
|
|
170
|
+
uri: direct.uri,
|
|
171
|
+
mimeType: direct.mime_type || 'video/mp4',
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
const steps = interaction.steps ?? []
|
|
175
|
+
for (let i = steps.length - 1; i >= 0; i--) {
|
|
176
|
+
const step = steps[i]
|
|
177
|
+
if (step?.type !== 'model_output') continue
|
|
178
|
+
for (const block of step.content ?? []) {
|
|
179
|
+
if (block.type === 'video' && (block.data || block.uri)) {
|
|
180
|
+
return {
|
|
181
|
+
data: block.data,
|
|
182
|
+
uri: block.uri,
|
|
183
|
+
mimeType: block.mime_type || 'video/mp4',
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
return undefined
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/**
|
|
192
|
+
* Map Interactions usage onto the canonical TokenUsage shape. Omni reports
|
|
193
|
+
* video output via `output_tokens_by_modality`; fall back to the video
|
|
194
|
+
* modality entry when the total is absent.
|
|
195
|
+
*/
|
|
196
|
+
function interactionUsageToTokenUsage(
|
|
197
|
+
usage: Interaction['usage'],
|
|
198
|
+
): TokenUsage | undefined {
|
|
199
|
+
if (!usage) return undefined
|
|
200
|
+
const videoTokens = usage.output_tokens_by_modality?.find(
|
|
201
|
+
(entry) => entry.modality === 'video',
|
|
202
|
+
)?.tokens
|
|
203
|
+
const promptTokens = usage.total_input_tokens ?? 0
|
|
204
|
+
const completionTokens = usage.total_output_tokens ?? videoTokens ?? 0
|
|
205
|
+
return {
|
|
206
|
+
promptTokens,
|
|
207
|
+
completionTokens,
|
|
208
|
+
totalTokens: usage.total_tokens ?? promptTokens + completionTokens,
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
/**
|
|
213
|
+
* Gemini Video Generation Adapter (Veo + Gemini Omni Flash)
|
|
103
214
|
*
|
|
104
|
-
* Tree-shakeable adapter for Google
|
|
105
|
-
* long-running operation: `createVideoJob` starts the operation via the
|
|
106
|
-
* `:predictLongRunning` endpoint, `getVideoStatus` polls it, and
|
|
107
|
-
* `getVideoUrl` extracts the generated video's URI once it completes.
|
|
215
|
+
* Tree-shakeable adapter for Google video generation, routing by model:
|
|
108
216
|
*
|
|
109
|
-
*
|
|
217
|
+
* **Veo models** run as a long-running operation: `createVideoJob` starts
|
|
218
|
+
* the operation via the `:predictLongRunning` endpoint, `getVideoStatus`
|
|
219
|
+
* polls it, and `getVideoUrl` extracts the generated video's URI once it
|
|
220
|
+
* completes. Image prompt parts are routed by `metadata.role`:
|
|
110
221
|
* - `'start_frame'` (or the first un-roled image) → the input image the
|
|
111
222
|
* video starts from
|
|
112
223
|
* - `'end_frame'` → `lastFrame` (the frame the video ends on)
|
|
113
224
|
* - `'reference'` / `'character'` → `referenceImages` (asset references,
|
|
114
225
|
* Veo 3.1)
|
|
115
226
|
*
|
|
116
|
-
* Note: the returned video URI is served by the Gemini Files API and
|
|
227
|
+
* Note: the returned Veo video URI is served by the Gemini Files API and
|
|
117
228
|
* requires the API key (`x-goog-api-key` header or `?key=` query
|
|
118
229
|
* parameter) to download.
|
|
119
230
|
*
|
|
231
|
+
* **Gemini Omni Flash** (`gemini-omni-flash-preview`) only serves the
|
|
232
|
+
* Interactions API: `createVideoJob` creates a background interaction with
|
|
233
|
+
* `response_modalities: ['video']`, `getVideoStatus` polls it by id, and
|
|
234
|
+
* `getVideoUrl` returns the inline base64 MP4 as a `data:` URL (or the
|
|
235
|
+
* Files API URI when the server delivers by reference). Image and video
|
|
236
|
+
* prompt parts are sent as interaction content blocks, grouped as images,
|
|
237
|
+
* then videos, then the text prompt (interleaving is not preserved); pass
|
|
238
|
+
* `modelOptions.previous_interaction_id` to conversationally edit a prior
|
|
239
|
+
* Omni generation.
|
|
240
|
+
*
|
|
120
241
|
* @experimental Video generation is an experimental feature and may change.
|
|
121
242
|
*/
|
|
122
243
|
export class GeminiVideoAdapter<
|
|
123
244
|
TModel extends GeminiVideoModel,
|
|
124
245
|
> extends BaseVideoAdapter<
|
|
125
246
|
TModel,
|
|
126
|
-
|
|
247
|
+
GeminiVideoModelProviderOptionsByName[TModel],
|
|
127
248
|
GeminiVideoModelProviderOptionsByName,
|
|
128
249
|
GeminiVideoModelSizeByName,
|
|
129
250
|
GeminiVideoModelInputModalitiesByName,
|
|
@@ -132,26 +253,35 @@ export class GeminiVideoAdapter<
|
|
|
132
253
|
readonly name = 'gemini' as const
|
|
133
254
|
|
|
134
255
|
protected client: GoogleGenAI
|
|
256
|
+
private readonly allowUrlFetch: boolean
|
|
135
257
|
|
|
136
258
|
constructor(config: GeminiVideoConfig, model: TModel) {
|
|
137
259
|
super({}, model)
|
|
138
260
|
this.client = createGeminiClient(config)
|
|
261
|
+
this.allowUrlFetch = config.allowUrlFetch ?? false
|
|
139
262
|
}
|
|
140
263
|
|
|
141
264
|
async createVideoJob(
|
|
142
265
|
options: VideoGenerationOptions<
|
|
143
|
-
|
|
266
|
+
GeminiVideoModelProviderOptionsByName[TModel],
|
|
144
267
|
GeminiVideoSize,
|
|
145
268
|
GeminiVideoModelDurationByName[TModel]
|
|
146
269
|
>,
|
|
147
270
|
): Promise<VideoJobResult> {
|
|
148
|
-
const { prompt, size, duration,
|
|
271
|
+
const { prompt, size, duration, logger } = options
|
|
149
272
|
|
|
150
273
|
logger.request(
|
|
151
274
|
`activity=video.create provider=${this.name} model=${this.model} size=${size ?? 'default'} duration=${duration ?? 'default'}`,
|
|
152
275
|
{ provider: this.name, model: this.model },
|
|
153
276
|
)
|
|
154
277
|
|
|
278
|
+
if (isInteractionsVideoModel(this.model)) {
|
|
279
|
+
return await this.createInteractionsVideoJob(options)
|
|
280
|
+
}
|
|
281
|
+
const modelOptions = options.modelOptions as
|
|
282
|
+
| GeminiVideoProviderOptions
|
|
283
|
+
| undefined
|
|
284
|
+
|
|
155
285
|
try {
|
|
156
286
|
const resolved = resolveMediaPrompt(prompt)
|
|
157
287
|
|
|
@@ -201,6 +331,99 @@ export class GeminiVideoAdapter<
|
|
|
201
331
|
}
|
|
202
332
|
}
|
|
203
333
|
|
|
334
|
+
/**
|
|
335
|
+
* Gemini Omni Flash job creation via the Interactions API. Creates a
|
|
336
|
+
* background interaction requesting video output; the interaction id is
|
|
337
|
+
* the job id polled by `getVideoStatus` / `getVideoUrl`.
|
|
338
|
+
*/
|
|
339
|
+
private async createInteractionsVideoJob(
|
|
340
|
+
options: VideoGenerationOptions<
|
|
341
|
+
GeminiVideoModelProviderOptionsByName[TModel],
|
|
342
|
+
GeminiVideoSize,
|
|
343
|
+
GeminiVideoModelDurationByName[TModel]
|
|
344
|
+
>,
|
|
345
|
+
): Promise<VideoJobResult> {
|
|
346
|
+
const { prompt, size, duration, logger } = options
|
|
347
|
+
const modelOptions = options.modelOptions as
|
|
348
|
+
| GeminiOmniVideoProviderOptions
|
|
349
|
+
| undefined
|
|
350
|
+
|
|
351
|
+
try {
|
|
352
|
+
const resolved = resolveMediaPrompt(prompt)
|
|
353
|
+
|
|
354
|
+
if (resolved.audios.length > 0) {
|
|
355
|
+
throw new Error(
|
|
356
|
+
`${this.name}.createVideoJob does not support audio prompt parts (model: ${this.model}).`,
|
|
357
|
+
)
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
const content: Array<InteractionContent> = [
|
|
361
|
+
...resolved.images.map(mediaPartToInteractionsContent),
|
|
362
|
+
...resolved.videos.map(mediaPartToInteractionsContent),
|
|
363
|
+
]
|
|
364
|
+
if (resolved.text) {
|
|
365
|
+
content.push({ type: 'text', text: resolved.text })
|
|
366
|
+
}
|
|
367
|
+
if (content.length === 0) {
|
|
368
|
+
throw new Error(
|
|
369
|
+
`${this.name}.createVideoJob: the prompt produced no content to send (model: ${this.model}).`,
|
|
370
|
+
)
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
// Reject out-of-range durations locally rather than snapping (which
|
|
374
|
+
// would silently change the clip length the caller asked for) or
|
|
375
|
+
// letting the live API reject them after the round trip.
|
|
376
|
+
const durations = this.availableDurations()
|
|
377
|
+
if (
|
|
378
|
+
duration !== undefined &&
|
|
379
|
+
durations.kind === 'range' &&
|
|
380
|
+
(duration < durations.min || duration > durations.max)
|
|
381
|
+
) {
|
|
382
|
+
throw new Error(
|
|
383
|
+
`${this.name}.createVideoJob: duration ${duration}s is outside the ${durations.min}–${durations.max}s range supported by ${this.model}. Use snapDuration() to snap arbitrary values into range.`,
|
|
384
|
+
)
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
// Aspect ratio and clip length ride on `response_format`. Duration is
|
|
388
|
+
// a `"<seconds>s"` string, accepted anywhere in the 3–10s range
|
|
389
|
+
// (fractional included) and defaulting to 10s when omitted — verified
|
|
390
|
+
// against the live API; the docs don't publish the range constraints.
|
|
391
|
+
const responseFormat =
|
|
392
|
+
size !== undefined || duration !== undefined
|
|
393
|
+
? {
|
|
394
|
+
response_format: {
|
|
395
|
+
type: 'video' as const,
|
|
396
|
+
...(size !== undefined && { aspect_ratio: size }),
|
|
397
|
+
...(duration !== undefined && { duration: `${duration}s` }),
|
|
398
|
+
},
|
|
399
|
+
}
|
|
400
|
+
: {}
|
|
401
|
+
|
|
402
|
+
const interaction = await this.client.interactions.create({
|
|
403
|
+
...modelOptions,
|
|
404
|
+
model: this.model,
|
|
405
|
+
input: [{ type: 'user_input', content }],
|
|
406
|
+
response_modalities: ['video'],
|
|
407
|
+
background: true,
|
|
408
|
+
...responseFormat,
|
|
409
|
+
})
|
|
410
|
+
|
|
411
|
+
if (!interaction.id) {
|
|
412
|
+
throw new Error(
|
|
413
|
+
'Gemini Omni did not return an interaction id for the video generation job.',
|
|
414
|
+
)
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
return { jobId: interaction.id, model: this.model }
|
|
418
|
+
} catch (error) {
|
|
419
|
+
logger.errors(`${this.name}.createVideoJob fatal`, {
|
|
420
|
+
error,
|
|
421
|
+
source: `${this.name}.createVideoJob`,
|
|
422
|
+
})
|
|
423
|
+
throw error
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
|
|
204
427
|
/**
|
|
205
428
|
* Route image prompt parts onto Veo's request fields by `metadata.role`.
|
|
206
429
|
*/
|
|
@@ -224,13 +447,13 @@ export class GeminiVideoAdapter<
|
|
|
224
447
|
`${this.name}: Veo accepts at most one 'end_frame' image.`,
|
|
225
448
|
)
|
|
226
449
|
}
|
|
227
|
-
lastFrame = await imagePartToVeoImage(part)
|
|
450
|
+
lastFrame = await imagePartToVeoImage(part, this.allowUrlFetch)
|
|
228
451
|
break
|
|
229
452
|
}
|
|
230
453
|
case 'reference':
|
|
231
454
|
case 'character': {
|
|
232
455
|
referenceImages.push({
|
|
233
|
-
image: await imagePartToVeoImage(part),
|
|
456
|
+
image: await imagePartToVeoImage(part, this.allowUrlFetch),
|
|
234
457
|
referenceType: VideoGenerationReferenceType.ASSET,
|
|
235
458
|
})
|
|
236
459
|
break
|
|
@@ -242,7 +465,7 @@ export class GeminiVideoAdapter<
|
|
|
242
465
|
`${this.name}: Veo accepts at most one starting image; received multiple 'start_frame'/un-roled images. Use metadata.role ('end_frame', 'reference') to disambiguate the others.`,
|
|
243
466
|
)
|
|
244
467
|
}
|
|
245
|
-
image = await imagePartToVeoImage(part)
|
|
468
|
+
image = await imagePartToVeoImage(part, this.allowUrlFetch)
|
|
246
469
|
break
|
|
247
470
|
}
|
|
248
471
|
case 'mask':
|
|
@@ -257,6 +480,9 @@ export class GeminiVideoAdapter<
|
|
|
257
480
|
}
|
|
258
481
|
|
|
259
482
|
async getVideoStatus(jobId: string): Promise<VideoStatusResult> {
|
|
483
|
+
if (isInteractionsVideoModel(this.model)) {
|
|
484
|
+
return await this.getInteractionsVideoStatus(jobId)
|
|
485
|
+
}
|
|
260
486
|
const operation = await this.getOperation(jobId)
|
|
261
487
|
|
|
262
488
|
if (!operation.done) {
|
|
@@ -289,7 +515,55 @@ export class GeminiVideoAdapter<
|
|
|
289
515
|
return { jobId, status: 'completed' }
|
|
290
516
|
}
|
|
291
517
|
|
|
518
|
+
/**
|
|
519
|
+
* Poll an Omni background interaction. `in_progress` maps to
|
|
520
|
+
* 'processing'; a `completed` interaction with no video content (e.g.
|
|
521
|
+
* filtered output) is surfaced as a failure so `getVideoUrl` doesn't
|
|
522
|
+
* throw on an empty response. `requires_action` also fails: the adapter
|
|
523
|
+
* never sends tools, so it can only arise via
|
|
524
|
+
* `previous_interaction_id` chaining onto a tool-bearing interaction —
|
|
525
|
+
* and such an interaction never progresses without a client response,
|
|
526
|
+
* so polling it would spin until timeout.
|
|
527
|
+
*/
|
|
528
|
+
private async getInteractionsVideoStatus(
|
|
529
|
+
jobId: string,
|
|
530
|
+
): Promise<VideoStatusResult> {
|
|
531
|
+
const interaction = await this.getInteraction(jobId)
|
|
532
|
+
const status = interaction.status
|
|
533
|
+
|
|
534
|
+
if (status === 'in_progress') {
|
|
535
|
+
return { jobId, status: 'processing' }
|
|
536
|
+
}
|
|
537
|
+
if (status === 'requires_action') {
|
|
538
|
+
return {
|
|
539
|
+
jobId,
|
|
540
|
+
status: 'failed',
|
|
541
|
+
error:
|
|
542
|
+
'Gemini Omni interaction is waiting on a client action (tool response), which the video jobs flow does not support.',
|
|
543
|
+
}
|
|
544
|
+
}
|
|
545
|
+
if (status === 'completed') {
|
|
546
|
+
if (!extractInteractionVideo(interaction)) {
|
|
547
|
+
return {
|
|
548
|
+
jobId,
|
|
549
|
+
status: 'failed',
|
|
550
|
+
error:
|
|
551
|
+
'Gemini Omni completed the interaction without returning a video (the output may have been filtered).',
|
|
552
|
+
}
|
|
553
|
+
}
|
|
554
|
+
return { jobId, status: 'completed' }
|
|
555
|
+
}
|
|
556
|
+
return {
|
|
557
|
+
jobId,
|
|
558
|
+
status: 'failed',
|
|
559
|
+
error: `Gemini Omni video generation ended with status "${status}".`,
|
|
560
|
+
}
|
|
561
|
+
}
|
|
562
|
+
|
|
292
563
|
async getVideoUrl(jobId: string): Promise<VideoUrlResult> {
|
|
564
|
+
if (isInteractionsVideoModel(this.model)) {
|
|
565
|
+
return await this.getInteractionsVideoUrl(jobId)
|
|
566
|
+
}
|
|
293
567
|
const operation = await this.getOperation(jobId)
|
|
294
568
|
|
|
295
569
|
if (!operation.done) {
|
|
@@ -317,6 +591,42 @@ export class GeminiVideoAdapter<
|
|
|
317
591
|
return { jobId, url: uri }
|
|
318
592
|
}
|
|
319
593
|
|
|
594
|
+
/**
|
|
595
|
+
* Extract the finished Omni video. Inline base64 output (the API default)
|
|
596
|
+
* becomes a `data:` URL — matching the OpenAI Sora adapter's inline
|
|
597
|
+
* delivery — and URI delivery passes through (Files API URIs need the API
|
|
598
|
+
* key to download, like Veo). Usage carries the video-modality output
|
|
599
|
+
* tokens (Omni bills per second of video, reported as tokens).
|
|
600
|
+
*/
|
|
601
|
+
private async getInteractionsVideoUrl(
|
|
602
|
+
jobId: string,
|
|
603
|
+
): Promise<VideoUrlResult> {
|
|
604
|
+
const interaction = await this.getInteraction(jobId)
|
|
605
|
+
const status = interaction.status
|
|
606
|
+
|
|
607
|
+
if (status === 'in_progress') {
|
|
608
|
+
throw new Error(
|
|
609
|
+
`Video is not ready yet. Check status first. Job ID: ${jobId}`,
|
|
610
|
+
)
|
|
611
|
+
}
|
|
612
|
+
if (status !== 'completed') {
|
|
613
|
+
throw new Error(
|
|
614
|
+
`Video generation failed: Gemini Omni interaction ended with status "${status}". Job ID: ${jobId}`,
|
|
615
|
+
)
|
|
616
|
+
}
|
|
617
|
+
|
|
618
|
+
const video = extractInteractionVideo(interaction)
|
|
619
|
+
if (!video) {
|
|
620
|
+
throw new Error(
|
|
621
|
+
`Video not found in interaction response (the output may have been filtered). Job ID: ${jobId}`,
|
|
622
|
+
)
|
|
623
|
+
}
|
|
624
|
+
|
|
625
|
+
const usage = interactionUsageToTokenUsage(interaction.usage)
|
|
626
|
+
const url = video.uri ?? `data:${video.mimeType};base64,${video.data}`
|
|
627
|
+
return { jobId, url, ...(usage && { usage }) }
|
|
628
|
+
}
|
|
629
|
+
|
|
320
630
|
override availableDurations(): DurationOptions<
|
|
321
631
|
GeminiVideoModelDurationByName[TModel]
|
|
322
632
|
> {
|
|
@@ -340,6 +650,13 @@ export class GeminiVideoAdapter<
|
|
|
340
650
|
operation.name = jobId
|
|
341
651
|
return await this.client.operations.getVideosOperation({ operation })
|
|
342
652
|
}
|
|
653
|
+
|
|
654
|
+
/**
|
|
655
|
+
* Fetch an Omni background interaction by id.
|
|
656
|
+
*/
|
|
657
|
+
private async getInteraction(jobId: string): Promise<Interaction> {
|
|
658
|
+
return await this.client.interactions.get(jobId)
|
|
659
|
+
}
|
|
343
660
|
}
|
|
344
661
|
|
|
345
662
|
/**
|
|
@@ -5,7 +5,7 @@ import {
|
|
|
5
5
|
createGeminiClient,
|
|
6
6
|
generateId,
|
|
7
7
|
getGeminiApiKeyFromEnv,
|
|
8
|
-
} from '../../utils'
|
|
8
|
+
} from '../../utils/client'
|
|
9
9
|
import type { InternalLogger } from '@tanstack/ai/adapter-internals'
|
|
10
10
|
import type {
|
|
11
11
|
GeminiChatModelToolCapabilitiesByName,
|
|
@@ -33,7 +33,7 @@ import type {
|
|
|
33
33
|
} from './events'
|
|
34
34
|
import type { ExternalTextInteractionsProviderOptions } from './provider-options'
|
|
35
35
|
import type { GeminiMessageMetadataByModality } from '../../message-types'
|
|
36
|
-
import type { GeminiClientConfig } from '../../utils'
|
|
36
|
+
import type { GeminiClientConfig } from '../../utils/client'
|
|
37
37
|
|
|
38
38
|
type Interaction = Interactions.Interaction
|
|
39
39
|
type InteractionSSEEvent = Interactions.InteractionSSEEvent
|