@tanstack/ai-gemini 0.16.2 → 0.17.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/image.d.ts +15 -2
- package/dist/esm/adapters/image.js +96 -8
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/adapters/video.d.ts +108 -0
- package/dist/esm/adapters/video.js +227 -0
- package/dist/esm/adapters/video.js.map +1 -0
- package/dist/esm/image/image-provider-options.d.ts +11 -0
- package/dist/esm/image/image-provider-options.js.map +1 -1
- package/dist/esm/index.d.ts +7 -0
- package/dist/esm/index.js +10 -2
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/model-meta.d.ts +5 -114
- package/dist/esm/model-meta.js +24 -1
- package/dist/esm/model-meta.js.map +1 -1
- package/dist/esm/video/video-provider-options.d.ts +90 -0
- package/dist/esm/video/video-provider-options.js +15 -0
- package/dist/esm/video/video-provider-options.js.map +1 -0
- package/package.json +3 -3
- package/src/adapters/image.ts +133 -14
- package/src/adapters/video.ts +411 -0
- package/src/image/image-provider-options.ts +12 -0
- package/src/index.ts +25 -0
- package/src/model-meta.ts +27 -26
- package/src/video/video-provider-options.ts +126 -0
package/src/adapters/image.ts
CHANGED
|
@@ -1,4 +1,6 @@
|
|
|
1
|
+
import { resolveMediaPrompt } from '@tanstack/ai'
|
|
1
2
|
import { BaseImageAdapter } from '@tanstack/ai/adapters'
|
|
3
|
+
import { arrayBufferToBase64 } from '@tanstack/ai-utils'
|
|
2
4
|
import {
|
|
3
5
|
createGeminiClient,
|
|
4
6
|
generateId,
|
|
@@ -14,6 +16,7 @@ import {
|
|
|
14
16
|
} from '../image/image-provider-options'
|
|
15
17
|
import type { GEMINI_IMAGE_MODELS } from '../model-meta'
|
|
16
18
|
import type {
|
|
19
|
+
GeminiImageModelInputModalitiesByName,
|
|
17
20
|
GeminiImageModelProviderOptionsByName,
|
|
18
21
|
GeminiImageModelSizeByName,
|
|
19
22
|
GeminiImageProviderOptions,
|
|
@@ -22,13 +25,18 @@ import type {
|
|
|
22
25
|
GeneratedImage,
|
|
23
26
|
ImageGenerationOptions,
|
|
24
27
|
ImageGenerationResult,
|
|
28
|
+
ImagePart,
|
|
29
|
+
MediaInputMetadata,
|
|
30
|
+
ResolvedMediaPrompt,
|
|
25
31
|
} from '@tanstack/ai'
|
|
26
32
|
import type {
|
|
33
|
+
Content,
|
|
27
34
|
GenerateContentConfig,
|
|
28
35
|
GenerateContentResponse,
|
|
29
36
|
GenerateImagesConfig,
|
|
30
37
|
GenerateImagesResponse,
|
|
31
38
|
GoogleGenAI,
|
|
39
|
+
Part,
|
|
32
40
|
} from '@google/genai'
|
|
33
41
|
import type { GeminiClientConfig } from '../utils'
|
|
34
42
|
|
|
@@ -60,7 +68,8 @@ export class GeminiImageAdapter<
|
|
|
60
68
|
TModel,
|
|
61
69
|
GeminiImageProviderOptions,
|
|
62
70
|
GeminiImageModelProviderOptionsByName,
|
|
63
|
-
GeminiImageModelSizeByName
|
|
71
|
+
GeminiImageModelSizeByName,
|
|
72
|
+
GeminiImageModelInputModalitiesByName
|
|
64
73
|
> {
|
|
65
74
|
override readonly kind = 'image' as const
|
|
66
75
|
readonly name = 'gemini' as const
|
|
@@ -70,6 +79,7 @@ export class GeminiImageAdapter<
|
|
|
70
79
|
providerOptions: GeminiImageProviderOptions
|
|
71
80
|
modelProviderOptionsByName: GeminiImageModelProviderOptionsByName
|
|
72
81
|
modelSizeByName: GeminiImageModelSizeByName
|
|
82
|
+
modelInputModalitiesByName: GeminiImageModelInputModalitiesByName
|
|
73
83
|
}
|
|
74
84
|
|
|
75
85
|
private readonly client: GoogleGenAI
|
|
@@ -82,7 +92,7 @@ export class GeminiImageAdapter<
|
|
|
82
92
|
async generateImages(
|
|
83
93
|
options: ImageGenerationOptions<GeminiImageProviderOptions>,
|
|
84
94
|
): Promise<ImageGenerationResult> {
|
|
85
|
-
const { model,
|
|
95
|
+
const { model, logger } = options
|
|
86
96
|
|
|
87
97
|
logger.request(
|
|
88
98
|
`activity=generateImage provider=gemini model=${this.model}`,
|
|
@@ -93,10 +103,35 @@ export class GeminiImageAdapter<
|
|
|
93
103
|
)
|
|
94
104
|
|
|
95
105
|
try {
|
|
96
|
-
|
|
106
|
+
const resolved = resolveMediaPrompt(options.prompt)
|
|
107
|
+
|
|
108
|
+
// Image-only prompts are allowed (the image inputs carry the intent);
|
|
109
|
+
// a prompt with neither text nor images is always an error.
|
|
110
|
+
if (resolved.images.length === 0) {
|
|
111
|
+
validatePrompt({ prompt: resolved.text, model })
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
if (resolved.videos.length > 0) {
|
|
115
|
+
throw new Error(
|
|
116
|
+
`${this.name}.generateImages does not support video prompt parts (model: ${model}).`,
|
|
117
|
+
)
|
|
118
|
+
}
|
|
119
|
+
if (resolved.audios.length > 0) {
|
|
120
|
+
throw new Error(
|
|
121
|
+
`${this.name}.generateImages does not support audio prompt parts (model: ${model}).`,
|
|
122
|
+
)
|
|
123
|
+
}
|
|
97
124
|
|
|
98
125
|
if (this.isGeminiImageModel(model)) {
|
|
99
|
-
return await this.generateWithGeminiApi(options)
|
|
126
|
+
return await this.generateWithGeminiApi(options, resolved)
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
// Imagen does not accept image inputs — it's strictly text-to-image.
|
|
130
|
+
if (resolved.images.length > 0) {
|
|
131
|
+
throw new Error(
|
|
132
|
+
`${this.name}: model "${model}" (Imagen) does not support image prompt parts. ` +
|
|
133
|
+
`Use a Gemini-native image model (e.g. gemini-2.5-flash-image, "nano-banana") for image-conditioned generation.`,
|
|
134
|
+
)
|
|
100
135
|
}
|
|
101
136
|
|
|
102
137
|
// Imagen models path (generateImages API)
|
|
@@ -107,7 +142,7 @@ export class GeminiImageAdapter<
|
|
|
107
142
|
|
|
108
143
|
const response = await this.client.models.generateImages({
|
|
109
144
|
model,
|
|
110
|
-
prompt,
|
|
145
|
+
prompt: resolved.text,
|
|
111
146
|
config,
|
|
112
147
|
})
|
|
113
148
|
|
|
@@ -127,18 +162,12 @@ export class GeminiImageAdapter<
|
|
|
127
162
|
|
|
128
163
|
private async generateWithGeminiApi(
|
|
129
164
|
options: ImageGenerationOptions<GeminiImageProviderOptions>,
|
|
165
|
+
resolved: ResolvedMediaPrompt,
|
|
130
166
|
): Promise<ImageGenerationResult> {
|
|
131
|
-
const { model,
|
|
167
|
+
const { model, size, numberOfImages, modelOptions } = options
|
|
132
168
|
|
|
133
169
|
const parsedSize = size ? parseNativeImageSize(size) : undefined
|
|
134
170
|
|
|
135
|
-
// The generateContent API has no numberOfImages parameter.
|
|
136
|
-
// Instead, augment the prompt to request multiple images when needed.
|
|
137
|
-
const augmentedPrompt =
|
|
138
|
-
numberOfImages && numberOfImages > 1
|
|
139
|
-
? `${prompt} Generate ${numberOfImages} distinct images.`
|
|
140
|
-
: prompt
|
|
141
|
-
|
|
142
171
|
// GeminiImageProviderOptions is Imagen-shaped — most fields
|
|
143
172
|
// (personGeneration, safetyFilterLevel, addWatermark, outputMimeType,
|
|
144
173
|
// outputCompressionQuality, guidanceScale, enhancePrompt,
|
|
@@ -170,15 +199,105 @@ export class GeminiImageAdapter<
|
|
|
170
199
|
}),
|
|
171
200
|
}
|
|
172
201
|
|
|
202
|
+
const contents = await this.buildContents(resolved, numberOfImages)
|
|
203
|
+
|
|
173
204
|
const response = await this.client.models.generateContent({
|
|
174
205
|
model,
|
|
175
|
-
contents
|
|
206
|
+
contents,
|
|
176
207
|
config,
|
|
177
208
|
})
|
|
178
209
|
|
|
179
210
|
return this.transformGeminiResponse(model, response)
|
|
180
211
|
}
|
|
181
212
|
|
|
213
|
+
/**
|
|
214
|
+
* Build the multimodal `contents` payload. Text-only prompts pass through
|
|
215
|
+
* as a plain string (the SDK accepts it directly); prompts with image
|
|
216
|
+
* parts become a single user `Content` whose `parts` mirror the prompt's
|
|
217
|
+
* interleaved order — position is meaningful to Gemini ("not like this
|
|
218
|
+
* *(image)*, more like this *(image)*").
|
|
219
|
+
*
|
|
220
|
+
* The generateContent API has no numberOfImages parameter, so when more
|
|
221
|
+
* than one image is requested a trailing instruction is appended.
|
|
222
|
+
*/
|
|
223
|
+
private async buildContents(
|
|
224
|
+
resolved: ResolvedMediaPrompt,
|
|
225
|
+
numberOfImages: number | undefined,
|
|
226
|
+
): Promise<string | Array<Content>> {
|
|
227
|
+
const countInstruction =
|
|
228
|
+
numberOfImages && numberOfImages > 1
|
|
229
|
+
? `Generate ${numberOfImages} distinct images.`
|
|
230
|
+
: undefined
|
|
231
|
+
|
|
232
|
+
if (resolved.images.length === 0) {
|
|
233
|
+
return countInstruction
|
|
234
|
+
? `${resolved.text} ${countInstruction}`
|
|
235
|
+
: resolved.text
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
const parts: Array<Part> = await Promise.all(
|
|
239
|
+
resolved.parts.map((part) => {
|
|
240
|
+
if (part.type === 'text') {
|
|
241
|
+
return Promise.resolve<Part>({ text: part.content })
|
|
242
|
+
}
|
|
243
|
+
if (part.type === 'image') {
|
|
244
|
+
return this.imagePartToGeminiPart(part)
|
|
245
|
+
}
|
|
246
|
+
// Video / audio parts were rejected in generateImages above.
|
|
247
|
+
throw new Error(
|
|
248
|
+
`gemini: unsupported prompt part type "${part.type}" in image generation.`,
|
|
249
|
+
)
|
|
250
|
+
}),
|
|
251
|
+
)
|
|
252
|
+
if (countInstruction) {
|
|
253
|
+
parts.push({ text: countInstruction })
|
|
254
|
+
}
|
|
255
|
+
return [{ role: 'user', parts }]
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
private async imagePartToGeminiPart(
|
|
259
|
+
part: ImagePart<MediaInputMetadata>,
|
|
260
|
+
): Promise<Part> {
|
|
261
|
+
if (part.source.type === 'data') {
|
|
262
|
+
return {
|
|
263
|
+
inlineData: {
|
|
264
|
+
mimeType: part.source.mimeType || 'image/png',
|
|
265
|
+
data: part.source.value,
|
|
266
|
+
},
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
// For URL sources, prefer passing the URL through as `fileData` when it
|
|
270
|
+
// looks like a Google Files API URI; otherwise fetch and inline as base64.
|
|
271
|
+
if (
|
|
272
|
+
part.source.value.startsWith('gs://') ||
|
|
273
|
+
/^https?:\/\/generativelanguage\.googleapis\.com\//.test(
|
|
274
|
+
part.source.value,
|
|
275
|
+
)
|
|
276
|
+
) {
|
|
277
|
+
return {
|
|
278
|
+
fileData: {
|
|
279
|
+
fileUri: part.source.value,
|
|
280
|
+
...(part.source.mimeType && { mimeType: part.source.mimeType }),
|
|
281
|
+
},
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
const response = await fetch(part.source.value)
|
|
285
|
+
if (!response.ok) {
|
|
286
|
+
throw new Error(
|
|
287
|
+
`Failed to fetch image input (${response.status} ${response.statusText}): ${part.source.value}`,
|
|
288
|
+
)
|
|
289
|
+
}
|
|
290
|
+
const blob = await response.blob()
|
|
291
|
+
const buffer = await blob.arrayBuffer()
|
|
292
|
+
const base64 = arrayBufferToBase64(buffer)
|
|
293
|
+
return {
|
|
294
|
+
inlineData: {
|
|
295
|
+
mimeType: part.source.mimeType || blob.type || 'image/png',
|
|
296
|
+
data: base64,
|
|
297
|
+
},
|
|
298
|
+
}
|
|
299
|
+
}
|
|
300
|
+
|
|
182
301
|
private transformGeminiResponse(
|
|
183
302
|
model: string,
|
|
184
303
|
response: GenerateContentResponse,
|
|
@@ -0,0 +1,411 @@
|
|
|
1
|
+
import {
|
|
2
|
+
GenerateVideosOperation,
|
|
3
|
+
VideoGenerationReferenceType,
|
|
4
|
+
} from '@google/genai'
|
|
5
|
+
import { resolveMediaPrompt } from '@tanstack/ai'
|
|
6
|
+
import { BaseVideoAdapter, snapToDurationOption } from '@tanstack/ai/adapters'
|
|
7
|
+
import { arrayBufferToBase64 } from '@tanstack/ai-utils'
|
|
8
|
+
import { createGeminiClient, getGeminiApiKeyFromEnv } from '../utils'
|
|
9
|
+
import { getGeminiVideoDurationOptions } from '../video/video-provider-options'
|
|
10
|
+
import type { DurationOptions } from '@tanstack/ai/adapters'
|
|
11
|
+
import type {
|
|
12
|
+
ImagePart,
|
|
13
|
+
MediaInputMetadata,
|
|
14
|
+
VideoGenerationOptions,
|
|
15
|
+
VideoJobResult,
|
|
16
|
+
VideoStatusResult,
|
|
17
|
+
VideoUrlResult,
|
|
18
|
+
} from '@tanstack/ai'
|
|
19
|
+
import type {
|
|
20
|
+
GenerateVideosConfig,
|
|
21
|
+
GoogleGenAI,
|
|
22
|
+
Image,
|
|
23
|
+
VideoGenerationReferenceImage,
|
|
24
|
+
} from '@google/genai'
|
|
25
|
+
import type {
|
|
26
|
+
GeminiVideoModel,
|
|
27
|
+
GeminiVideoModelDurationByName,
|
|
28
|
+
GeminiVideoModelInputModalitiesByName,
|
|
29
|
+
GeminiVideoModelProviderOptionsByName,
|
|
30
|
+
GeminiVideoModelSizeByName,
|
|
31
|
+
GeminiVideoProviderOptions,
|
|
32
|
+
GeminiVideoSize,
|
|
33
|
+
} from '../video/video-provider-options'
|
|
34
|
+
import type { GeminiClientConfig } from '../utils'
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Configuration for Gemini video adapter.
|
|
38
|
+
*
|
|
39
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
40
|
+
*/
|
|
41
|
+
export interface GeminiVideoConfig extends GeminiClientConfig {}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Extract a human-readable message from a long-running operation's error,
|
|
45
|
+
* which the SDK types as `Record<string, unknown>` (a google.rpc.Status).
|
|
46
|
+
*/
|
|
47
|
+
function operationErrorMessage(error: Record<string, unknown>): string {
|
|
48
|
+
if (typeof error.message === 'string' && error.message.length > 0) {
|
|
49
|
+
return error.message
|
|
50
|
+
}
|
|
51
|
+
return JSON.stringify(error)
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* Convert a TanStack image prompt part into the genai `Image` shape Veo
|
|
56
|
+
* accepts: base64 `imageBytes` (data sources, data: URIs, fetched HTTP
|
|
57
|
+
* URLs) or a `gcsUri` passthrough for Cloud Storage references.
|
|
58
|
+
*/
|
|
59
|
+
async function imagePartToVeoImage(
|
|
60
|
+
part: ImagePart<MediaInputMetadata>,
|
|
61
|
+
): Promise<Image> {
|
|
62
|
+
if (part.source.type === 'data') {
|
|
63
|
+
return {
|
|
64
|
+
imageBytes: part.source.value,
|
|
65
|
+
mimeType: part.source.mimeType || 'image/png',
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
const url = part.source.value
|
|
69
|
+
if (url.startsWith('gs://')) {
|
|
70
|
+
return {
|
|
71
|
+
gcsUri: url,
|
|
72
|
+
...(part.source.mimeType && { mimeType: part.source.mimeType }),
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
if (url.startsWith('data:')) {
|
|
76
|
+
const match = url.match(/^data:([^;,]+)?(;base64)?,(.*)$/)
|
|
77
|
+
if (!match || !match[2]) {
|
|
78
|
+
throw new Error(
|
|
79
|
+
'gemini: only base64 data: URIs are supported for video image inputs.',
|
|
80
|
+
)
|
|
81
|
+
}
|
|
82
|
+
return {
|
|
83
|
+
imageBytes: match[3] ?? '',
|
|
84
|
+
mimeType: match[1] || part.source.mimeType || 'image/png',
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
const response = await fetch(url)
|
|
88
|
+
if (!response.ok) {
|
|
89
|
+
throw new Error(
|
|
90
|
+
`Failed to fetch image input (${response.status} ${response.statusText}): ${url}`,
|
|
91
|
+
)
|
|
92
|
+
}
|
|
93
|
+
const blob = await response.blob()
|
|
94
|
+
const buffer = await blob.arrayBuffer()
|
|
95
|
+
return {
|
|
96
|
+
imageBytes: arrayBufferToBase64(buffer),
|
|
97
|
+
mimeType: part.source.mimeType || blob.type || 'image/png',
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
/**
|
|
102
|
+
* Gemini Veo Video Generation Adapter
|
|
103
|
+
*
|
|
104
|
+
* Tree-shakeable adapter for Google Veo video generation. Veo runs as a
|
|
105
|
+
* long-running operation: `createVideoJob` starts the operation via the
|
|
106
|
+
* `:predictLongRunning` endpoint, `getVideoStatus` polls it, and
|
|
107
|
+
* `getVideoUrl` extracts the generated video's URI once it completes.
|
|
108
|
+
*
|
|
109
|
+
* Image prompt parts are routed by `metadata.role`:
|
|
110
|
+
* - `'start_frame'` (or the first un-roled image) → the input image the
|
|
111
|
+
* video starts from
|
|
112
|
+
* - `'end_frame'` → `lastFrame` (the frame the video ends on)
|
|
113
|
+
* - `'reference'` / `'character'` → `referenceImages` (asset references,
|
|
114
|
+
* Veo 3.1)
|
|
115
|
+
*
|
|
116
|
+
* Note: the returned video URI is served by the Gemini Files API and
|
|
117
|
+
* requires the API key (`x-goog-api-key` header or `?key=` query
|
|
118
|
+
* parameter) to download.
|
|
119
|
+
*
|
|
120
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
121
|
+
*/
|
|
122
|
+
export class GeminiVideoAdapter<
|
|
123
|
+
TModel extends GeminiVideoModel,
|
|
124
|
+
> extends BaseVideoAdapter<
|
|
125
|
+
TModel,
|
|
126
|
+
GeminiVideoProviderOptions,
|
|
127
|
+
GeminiVideoModelProviderOptionsByName,
|
|
128
|
+
GeminiVideoModelSizeByName,
|
|
129
|
+
GeminiVideoModelInputModalitiesByName,
|
|
130
|
+
GeminiVideoModelDurationByName
|
|
131
|
+
> {
|
|
132
|
+
readonly name = 'gemini' as const
|
|
133
|
+
|
|
134
|
+
protected client: GoogleGenAI
|
|
135
|
+
|
|
136
|
+
constructor(config: GeminiVideoConfig, model: TModel) {
|
|
137
|
+
super({}, model)
|
|
138
|
+
this.client = createGeminiClient(config)
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
async createVideoJob(
|
|
142
|
+
options: VideoGenerationOptions<
|
|
143
|
+
GeminiVideoProviderOptions,
|
|
144
|
+
GeminiVideoSize,
|
|
145
|
+
GeminiVideoModelDurationByName[TModel]
|
|
146
|
+
>,
|
|
147
|
+
): Promise<VideoJobResult> {
|
|
148
|
+
const { prompt, size, duration, modelOptions, logger } = options
|
|
149
|
+
|
|
150
|
+
logger.request(
|
|
151
|
+
`activity=video.create provider=${this.name} model=${this.model} size=${size ?? 'default'} duration=${duration ?? 'default'}`,
|
|
152
|
+
{ provider: this.name, model: this.model },
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
try {
|
|
156
|
+
const resolved = resolveMediaPrompt(prompt)
|
|
157
|
+
|
|
158
|
+
if (resolved.videos.length > 0) {
|
|
159
|
+
throw new Error(
|
|
160
|
+
`${this.name}.createVideoJob does not support video prompt parts (model: ${this.model}).`,
|
|
161
|
+
)
|
|
162
|
+
}
|
|
163
|
+
if (resolved.audios.length > 0) {
|
|
164
|
+
throw new Error(
|
|
165
|
+
`${this.name}.createVideoJob does not support audio prompt parts (model: ${this.model}).`,
|
|
166
|
+
)
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
const { image, lastFrame, referenceImages } = await this.routeImageParts(
|
|
170
|
+
resolved.images,
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
const config: GenerateVideosConfig = {
|
|
174
|
+
...modelOptions,
|
|
175
|
+
...(size !== undefined && { aspectRatio: size }),
|
|
176
|
+
...(duration !== undefined && { durationSeconds: duration }),
|
|
177
|
+
...(lastFrame && { lastFrame }),
|
|
178
|
+
...(referenceImages.length > 0 && { referenceImages }),
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
const operation = await this.client.models.generateVideos({
|
|
182
|
+
model: this.model,
|
|
183
|
+
prompt: resolved.text,
|
|
184
|
+
...(image && { image }),
|
|
185
|
+
config,
|
|
186
|
+
})
|
|
187
|
+
|
|
188
|
+
if (!operation.name) {
|
|
189
|
+
throw new Error(
|
|
190
|
+
'Veo did not return an operation name for the video generation job.',
|
|
191
|
+
)
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
return { jobId: operation.name, model: this.model }
|
|
195
|
+
} catch (error) {
|
|
196
|
+
logger.errors(`${this.name}.createVideoJob fatal`, {
|
|
197
|
+
error,
|
|
198
|
+
source: `${this.name}.createVideoJob`,
|
|
199
|
+
})
|
|
200
|
+
throw error
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* Route image prompt parts onto Veo's request fields by `metadata.role`.
|
|
206
|
+
*/
|
|
207
|
+
private async routeImageParts(
|
|
208
|
+
parts: Array<ImagePart<MediaInputMetadata>>,
|
|
209
|
+
): Promise<{
|
|
210
|
+
image: Image | undefined
|
|
211
|
+
lastFrame: Image | undefined
|
|
212
|
+
referenceImages: Array<VideoGenerationReferenceImage>
|
|
213
|
+
}> {
|
|
214
|
+
let image: Image | undefined
|
|
215
|
+
let lastFrame: Image | undefined
|
|
216
|
+
const referenceImages: Array<VideoGenerationReferenceImage> = []
|
|
217
|
+
|
|
218
|
+
for (const part of parts) {
|
|
219
|
+
const role = part.metadata?.role
|
|
220
|
+
switch (role) {
|
|
221
|
+
case 'end_frame': {
|
|
222
|
+
if (lastFrame) {
|
|
223
|
+
throw new Error(
|
|
224
|
+
`${this.name}: Veo accepts at most one 'end_frame' image.`,
|
|
225
|
+
)
|
|
226
|
+
}
|
|
227
|
+
lastFrame = await imagePartToVeoImage(part)
|
|
228
|
+
break
|
|
229
|
+
}
|
|
230
|
+
case 'reference':
|
|
231
|
+
case 'character': {
|
|
232
|
+
referenceImages.push({
|
|
233
|
+
image: await imagePartToVeoImage(part),
|
|
234
|
+
referenceType: VideoGenerationReferenceType.ASSET,
|
|
235
|
+
})
|
|
236
|
+
break
|
|
237
|
+
}
|
|
238
|
+
case 'start_frame':
|
|
239
|
+
case undefined: {
|
|
240
|
+
if (image) {
|
|
241
|
+
throw new Error(
|
|
242
|
+
`${this.name}: Veo accepts at most one starting image; received multiple 'start_frame'/un-roled images. Use metadata.role ('end_frame', 'reference') to disambiguate the others.`,
|
|
243
|
+
)
|
|
244
|
+
}
|
|
245
|
+
image = await imagePartToVeoImage(part)
|
|
246
|
+
break
|
|
247
|
+
}
|
|
248
|
+
case 'mask':
|
|
249
|
+
case 'control':
|
|
250
|
+
throw new Error(
|
|
251
|
+
`${this.name}: unsupported image role "${role}" for Veo video generation.`,
|
|
252
|
+
)
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
return { image, lastFrame, referenceImages }
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
async getVideoStatus(jobId: string): Promise<VideoStatusResult> {
|
|
260
|
+
const operation = await this.getOperation(jobId)
|
|
261
|
+
|
|
262
|
+
if (!operation.done) {
|
|
263
|
+
return { jobId, status: 'processing' }
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
if (operation.error) {
|
|
267
|
+
return {
|
|
268
|
+
jobId,
|
|
269
|
+
status: 'failed',
|
|
270
|
+
error: operationErrorMessage(operation.error),
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
// The operation can finish "successfully" with every sample dropped by
|
|
275
|
+
// Responsible-AI filters — surface that as a failure instead of letting
|
|
276
|
+
// getVideoUrl() throw on an empty response.
|
|
277
|
+
const videos = operation.response?.generatedVideos ?? []
|
|
278
|
+
if (videos.length === 0) {
|
|
279
|
+
const reasons = operation.response?.raiMediaFilteredReasons
|
|
280
|
+
return {
|
|
281
|
+
jobId,
|
|
282
|
+
status: 'failed',
|
|
283
|
+
error: reasons?.length
|
|
284
|
+
? `Video was filtered by Responsible-AI: ${reasons.join('; ')}`
|
|
285
|
+
: 'Veo returned no generated videos.',
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
return { jobId, status: 'completed' }
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
async getVideoUrl(jobId: string): Promise<VideoUrlResult> {
|
|
293
|
+
const operation = await this.getOperation(jobId)
|
|
294
|
+
|
|
295
|
+
if (!operation.done) {
|
|
296
|
+
throw new Error(
|
|
297
|
+
`Video is not ready yet. Check status first. Job ID: ${jobId}`,
|
|
298
|
+
)
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
if (operation.error) {
|
|
302
|
+
throw new Error(
|
|
303
|
+
`Video generation failed: ${operationErrorMessage(operation.error)}`,
|
|
304
|
+
)
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
const uri = operation.response?.generatedVideos?.[0]?.video?.uri
|
|
308
|
+
if (!uri) {
|
|
309
|
+
const reasons = operation.response?.raiMediaFilteredReasons
|
|
310
|
+
throw new Error(
|
|
311
|
+
reasons?.length
|
|
312
|
+
? `Video was filtered by Responsible-AI: ${reasons.join('; ')}`
|
|
313
|
+
: `Video URL not found in operation response. Job ID: ${jobId}`,
|
|
314
|
+
)
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
return { jobId, url: uri }
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
override availableDurations(): DurationOptions<
|
|
321
|
+
GeminiVideoModelDurationByName[TModel]
|
|
322
|
+
> {
|
|
323
|
+
return getGeminiVideoDurationOptions(this.model)
|
|
324
|
+
}
|
|
325
|
+
|
|
326
|
+
override snapDuration(
|
|
327
|
+
seconds: number,
|
|
328
|
+
): GeminiVideoModelDurationByName[TModel] | undefined {
|
|
329
|
+
return snapToDurationOption(seconds, this.availableDurations())
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
/**
|
|
333
|
+
* Fetch the long-running operation by name. The SDK's
|
|
334
|
+
* `operations.getVideosOperation` needs a real `GenerateVideosOperation`
|
|
335
|
+
* instance (it calls `_fromAPIResponse` on it), so reconstruct one from
|
|
336
|
+
* the job ID rather than passing an object literal.
|
|
337
|
+
*/
|
|
338
|
+
private async getOperation(jobId: string): Promise<GenerateVideosOperation> {
|
|
339
|
+
const operation = new GenerateVideosOperation()
|
|
340
|
+
operation.name = jobId
|
|
341
|
+
return await this.client.operations.getVideosOperation({ operation })
|
|
342
|
+
}
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
/**
|
|
346
|
+
* Creates a Gemini video adapter with an explicit API key.
|
|
347
|
+
* Type resolution happens here at the call site.
|
|
348
|
+
*
|
|
349
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
350
|
+
*
|
|
351
|
+
* @param model - The model name (e.g., 'veo-3.1-generate-preview')
|
|
352
|
+
* @param apiKey - Your Google API key
|
|
353
|
+
* @param config - Optional additional configuration
|
|
354
|
+
* @returns Configured Gemini video adapter instance with resolved types
|
|
355
|
+
*
|
|
356
|
+
* @example
|
|
357
|
+
* ```typescript
|
|
358
|
+
* const adapter = createGeminiVideo('veo-3.1-generate-preview', 'your-api-key');
|
|
359
|
+
*
|
|
360
|
+
* const { jobId } = await generateVideo({
|
|
361
|
+
* adapter,
|
|
362
|
+
* prompt: 'A beautiful sunset over the ocean',
|
|
363
|
+
* duration: adapter.snapDuration(7), // → 6
|
|
364
|
+
* });
|
|
365
|
+
* ```
|
|
366
|
+
*/
|
|
367
|
+
export function createGeminiVideo<TModel extends GeminiVideoModel>(
|
|
368
|
+
model: TModel,
|
|
369
|
+
apiKey: string,
|
|
370
|
+
config?: Omit<GeminiVideoConfig, 'apiKey'>,
|
|
371
|
+
): GeminiVideoAdapter<TModel> {
|
|
372
|
+
return new GeminiVideoAdapter({ apiKey, ...config }, model)
|
|
373
|
+
}
|
|
374
|
+
|
|
375
|
+
/**
|
|
376
|
+
* Creates a Gemini video adapter with automatic API key detection from environment variables.
|
|
377
|
+
* Type resolution happens here at the call site.
|
|
378
|
+
*
|
|
379
|
+
* Looks for `GOOGLE_API_KEY` or `GEMINI_API_KEY` in:
|
|
380
|
+
* - `process.env` (Node.js)
|
|
381
|
+
* - `window.env` (Browser with injected env)
|
|
382
|
+
*
|
|
383
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
384
|
+
*
|
|
385
|
+
* @param model - The model name (e.g., 'veo-3.1-generate-preview')
|
|
386
|
+
* @param config - Optional configuration (excluding apiKey which is auto-detected)
|
|
387
|
+
* @returns Configured Gemini video adapter instance with resolved types
|
|
388
|
+
* @throws Error if GOOGLE_API_KEY or GEMINI_API_KEY is not found in environment
|
|
389
|
+
*
|
|
390
|
+
* @example
|
|
391
|
+
* ```typescript
|
|
392
|
+
* // Automatically uses GOOGLE_API_KEY from environment
|
|
393
|
+
* const adapter = geminiVideo('veo-3.1-generate-preview');
|
|
394
|
+
*
|
|
395
|
+
* // Create a video generation job
|
|
396
|
+
* const { jobId } = await generateVideo({
|
|
397
|
+
* adapter,
|
|
398
|
+
* prompt: 'A cat playing piano'
|
|
399
|
+
* });
|
|
400
|
+
*
|
|
401
|
+
* // Poll for status
|
|
402
|
+
* const status = await getVideoJobStatus({ adapter, jobId });
|
|
403
|
+
* ```
|
|
404
|
+
*/
|
|
405
|
+
export function geminiVideo<TModel extends GeminiVideoModel>(
|
|
406
|
+
model: TModel,
|
|
407
|
+
config?: Omit<GeminiVideoConfig, 'apiKey'>,
|
|
408
|
+
): GeminiVideoAdapter<TModel> {
|
|
409
|
+
const apiKey = getGeminiApiKeyFromEnv()
|
|
410
|
+
return createGeminiVideo(model, apiKey, config)
|
|
411
|
+
}
|
|
@@ -189,6 +189,18 @@ export type GeminiImageModelSizeByName = {
|
|
|
189
189
|
[K in Exclude<GeminiImageModels, GeminiNativeImageModels>]: GeminiImageSize
|
|
190
190
|
}
|
|
191
191
|
|
|
192
|
+
/**
|
|
193
|
+
* Per-model prompt input modalities. Gemini-native image models accept image
|
|
194
|
+
* parts in the multimodal prompt (image-conditioned generation via
|
|
195
|
+
* generateContent); Imagen models are strictly text-to-image, so their
|
|
196
|
+
* `prompt` is constrained to text at compile time.
|
|
197
|
+
*/
|
|
198
|
+
export type GeminiImageModelInputModalitiesByName = {
|
|
199
|
+
[K in GeminiNativeImageModels]: readonly ['image']
|
|
200
|
+
} & {
|
|
201
|
+
[K in Exclude<GeminiImageModels, GeminiNativeImageModels>]: readonly []
|
|
202
|
+
}
|
|
203
|
+
|
|
192
204
|
/**
|
|
193
205
|
* Valid sizes for Gemini Imagen models
|
|
194
206
|
* Gemini uses aspect ratios, but we map common WIDTHxHEIGHT formats to aspect ratios
|
package/src/index.ts
CHANGED
|
@@ -61,6 +61,30 @@ export {
|
|
|
61
61
|
type GeminiAudioProviderOptions,
|
|
62
62
|
} from './adapters/audio'
|
|
63
63
|
|
|
64
|
+
// Video / Veo generation adapter (experimental)
|
|
65
|
+
/**
|
|
66
|
+
* @experimental Veo video generation is an experimental feature and may change.
|
|
67
|
+
*/
|
|
68
|
+
export {
|
|
69
|
+
GeminiVideoAdapter,
|
|
70
|
+
createGeminiVideo,
|
|
71
|
+
geminiVideo,
|
|
72
|
+
type GeminiVideoConfig,
|
|
73
|
+
} from './adapters/video'
|
|
74
|
+
export {
|
|
75
|
+
GEMINI_VIDEO_DURATIONS,
|
|
76
|
+
getGeminiVideoDurationOptions,
|
|
77
|
+
} from './video/video-provider-options'
|
|
78
|
+
export type {
|
|
79
|
+
GeminiVideoModel,
|
|
80
|
+
GeminiVideoModelDurationByName,
|
|
81
|
+
GeminiVideoModelInputModalitiesByName,
|
|
82
|
+
GeminiVideoModelProviderOptionsByName,
|
|
83
|
+
GeminiVideoModelSizeByName,
|
|
84
|
+
GeminiVideoProviderOptions,
|
|
85
|
+
GeminiVideoSize,
|
|
86
|
+
} from './video/video-provider-options'
|
|
87
|
+
|
|
64
88
|
// Re-export models from model-meta for convenience
|
|
65
89
|
export {
|
|
66
90
|
GEMINI_MODELS,
|
|
@@ -71,6 +95,7 @@ export { GEMINI_IMAGE_MODELS as GeminiImageModels } from './model-meta'
|
|
|
71
95
|
export { GEMINI_TTS_MODELS as GeminiTTSModels } from './model-meta'
|
|
72
96
|
export { GEMINI_TTS_VOICES as GeminiTTSVoices } from './model-meta'
|
|
73
97
|
export { GEMINI_AUDIO_MODELS as GeminiAudioModels } from './model-meta'
|
|
98
|
+
export { GEMINI_VIDEO_MODELS as GeminiVideoModels } from './model-meta'
|
|
74
99
|
export type { GeminiModels as GeminiTextModel } from './model-meta'
|
|
75
100
|
export type { GeminiImageModels as GeminiImageModel } from './model-meta'
|
|
76
101
|
export type { GeminiTTSVoice } from './model-meta'
|