@tanstack/ai-gemini 0.20.0 → 0.21.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/README.md +15 -1
  2. package/dist/esm/adapters/audio.js +113 -61
  3. package/dist/esm/adapters/audio.js.map +1 -1
  4. package/dist/esm/adapters/image.js +224 -225
  5. package/dist/esm/adapters/image.js.map +1 -1
  6. package/dist/esm/adapters/summarize.js +33 -12
  7. package/dist/esm/adapters/summarize.js.map +1 -1
  8. package/dist/esm/adapters/text.js +574 -650
  9. package/dist/esm/adapters/text.js.map +1 -1
  10. package/dist/esm/adapters/tts.js +197 -178
  11. package/dist/esm/adapters/tts.js.map +1 -1
  12. package/dist/esm/adapters/video.js +456 -409
  13. package/dist/esm/adapters/video.js.map +1 -1
  14. package/dist/esm/experimental/index.js +1 -6
  15. package/dist/esm/experimental/text-interactions/adapter.js +877 -1014
  16. package/dist/esm/experimental/text-interactions/adapter.js.map +1 -1
  17. package/dist/esm/image/image-provider-options.js +74 -54
  18. package/dist/esm/image/image-provider-options.js.map +1 -1
  19. package/dist/esm/index.js +4 -36
  20. package/dist/esm/model-meta.d.ts +53 -5
  21. package/dist/esm/model-meta.js +733 -129
  22. package/dist/esm/model-meta.js.map +1 -1
  23. package/dist/esm/realtime/adapter.js +242 -228
  24. package/dist/esm/realtime/adapter.js.map +1 -1
  25. package/dist/esm/realtime/client.js +314 -381
  26. package/dist/esm/realtime/client.js.map +1 -1
  27. package/dist/esm/realtime/index.js +3 -0
  28. package/dist/esm/realtime/token.js +50 -36
  29. package/dist/esm/realtime/token.js.map +1 -1
  30. package/dist/esm/realtime/utils.js +202 -245
  31. package/dist/esm/realtime/utils.js.map +1 -1
  32. package/dist/esm/tools/code-execution-tool.js +11 -13
  33. package/dist/esm/tools/code-execution-tool.js.map +1 -1
  34. package/dist/esm/tools/computer-use-tool.js +18 -28
  35. package/dist/esm/tools/computer-use-tool.js.map +1 -1
  36. package/dist/esm/tools/file-search-tool.js +11 -14
  37. package/dist/esm/tools/file-search-tool.js.map +1 -1
  38. package/dist/esm/tools/function-declaration-tool.js +11 -25
  39. package/dist/esm/tools/function-declaration-tool.js.map +1 -1
  40. package/dist/esm/tools/google-maps-tool.js +11 -14
  41. package/dist/esm/tools/google-maps-tool.js.map +1 -1
  42. package/dist/esm/tools/google-search-retriveal-tool.js +11 -14
  43. package/dist/esm/tools/google-search-retriveal-tool.js.map +1 -1
  44. package/dist/esm/tools/google-search-tool.js +11 -14
  45. package/dist/esm/tools/google-search-tool.js.map +1 -1
  46. package/dist/esm/tools/index.js +2 -13
  47. package/dist/esm/tools/tool-converter.js +64 -57
  48. package/dist/esm/tools/tool-converter.js.map +1 -1
  49. package/dist/esm/tools/url-context-tool.js +11 -13
  50. package/dist/esm/tools/url-context-tool.js.map +1 -1
  51. package/dist/esm/usage.js +81 -87
  52. package/dist/esm/usage.js.map +1 -1
  53. package/dist/esm/utils/client.js +31 -24
  54. package/dist/esm/utils/client.js.map +1 -1
  55. package/dist/esm/utils/index.js +2 -0
  56. package/dist/esm/video/video-provider-options.js +73 -20
  57. package/dist/esm/video/video-provider-options.js.map +1 -1
  58. package/package.json +7 -7
  59. package/src/model-meta.ts +120 -10
  60. package/dist/esm/experimental/index.js.map +0 -1
  61. package/dist/esm/index.js.map +0 -1
  62. package/dist/esm/tools/index.js.map +0 -1
@@ -1,426 +1,473 @@
1
- import { VideoGenerationReferenceType, GenerateVideosOperation } from "@google/genai";
1
+ import { createGeminiClient, getGeminiApiKeyFromEnv } from "../utils/client.js";
2
+ import "../utils/index.js";
3
+ import { getGeminiVideoDurationOptions, isInteractionsVideoModel } from "../video/video-provider-options.js";
4
+ import { GenerateVideosOperation, VideoGenerationReferenceType } from "@google/genai";
2
5
  import { resolveMediaPrompt } from "@tanstack/ai";
3
6
  import { BaseVideoAdapter, snapToDurationOption } from "@tanstack/ai/adapters";
4
7
  import { arrayBufferToBase64 } from "@tanstack/ai-utils";
5
- import { createGeminiClient, getGeminiApiKeyFromEnv } from "../utils/client.js";
6
- import { isInteractionsVideoModel, getGeminiVideoDurationOptions } from "../video/video-provider-options.js";
8
+ //#region src/adapters/video.ts
9
+ /**
10
+ * Extract a human-readable message from a long-running operation's error,
11
+ * which the SDK types as `Record<string, unknown>` (a google.rpc.Status).
12
+ */
7
13
  function operationErrorMessage(error) {
8
- if (typeof error.message === "string" && error.message.length > 0) {
9
- return error.message;
10
- }
11
- return JSON.stringify(error);
14
+ if (typeof error.message === "string" && error.message.length > 0) return error.message;
15
+ return JSON.stringify(error);
12
16
  }
17
+ /**
18
+ * Convert a TanStack image prompt part into the genai `Image` shape Veo
19
+ * accepts: base64 `imageBytes` (data sources, data: URIs, fetched HTTP
20
+ * URLs) or a `gcsUri` passthrough for Cloud Storage references.
21
+ *
22
+ * Unlike `generateContent` (chat / native image generation), Veo's predict
23
+ * API has no `fileData.fileUri` equivalent — `Image` only accepts
24
+ * `imageBytes` or `gcsUri`. An HTTP(S) URL therefore has to be fetched and
25
+ * inlined locally, which buffers the whole image in memory; that only happens
26
+ * when the caller opts in via `allowUrlFetch`, otherwise it throws. Prefer a
27
+ * `gs://` reference on memory-constrained runtimes.
28
+ */
13
29
  async function imagePartToVeoImage(part, allowUrlFetch) {
14
- if (part.source.type === "data") {
15
- return {
16
- imageBytes: part.source.value,
17
- mimeType: part.source.mimeType || "image/png"
18
- };
19
- }
20
- const url = part.source.value;
21
- if (url.startsWith("gs://")) {
22
- return {
23
- gcsUri: url,
24
- ...part.source.mimeType && { mimeType: part.source.mimeType }
25
- };
26
- }
27
- if (url.startsWith("data:")) {
28
- const match = url.match(/^data:([^;,]+)?(;base64)?,(.*)$/);
29
- if (!match || !match[2]) {
30
- throw new Error(
31
- "gemini: only base64 data: URIs are supported for video image inputs."
32
- );
33
- }
34
- return {
35
- imageBytes: match[3] ?? "",
36
- mimeType: match[1] || part.source.mimeType || "image/png"
37
- };
38
- }
39
- if (!allowUrlFetch) {
40
- throw new Error(
41
- `gemini Veo: HTTP(S) URL image inputs are not fetched by default because Veo accepts only inline bytes, so the image would be downloaded and buffered in memory (risking OOM on constrained runtimes). Pass a data: URI or a gs:// reference, or set \`allowUrlFetch: true\` on the adapter config to opt into fetching. URL: ${url}`
42
- );
43
- }
44
- const response = await fetch(url);
45
- if (!response.ok) {
46
- throw new Error(
47
- `Failed to fetch image input (${response.status} ${response.statusText}): ${url}`
48
- );
49
- }
50
- const blob = await response.blob();
51
- const buffer = await blob.arrayBuffer();
52
- return {
53
- imageBytes: arrayBufferToBase64(buffer),
54
- mimeType: part.source.mimeType || blob.type || "image/png"
55
- };
30
+ if (part.source.type === "data") return {
31
+ imageBytes: part.source.value,
32
+ mimeType: part.source.mimeType || "image/png"
33
+ };
34
+ const url = part.source.value;
35
+ if (url.startsWith("gs://")) return {
36
+ gcsUri: url,
37
+ ...part.source.mimeType && { mimeType: part.source.mimeType }
38
+ };
39
+ if (url.startsWith("data:")) {
40
+ const match = url.match(/^data:([^;,]+)?(;base64)?,(.*)$/);
41
+ if (!match || !match[2]) throw new Error("gemini: only base64 data: URIs are supported for video image inputs.");
42
+ return {
43
+ imageBytes: match[3] ?? "",
44
+ mimeType: match[1] || part.source.mimeType || "image/png"
45
+ };
46
+ }
47
+ if (!allowUrlFetch) throw new Error(`gemini Veo: HTTP(S) URL image inputs are not fetched by default because Veo accepts only inline bytes, so the image would be downloaded and buffered in memory (risking OOM on constrained runtimes). Pass a data: URI or a gs:// reference, or set \`allowUrlFetch: true\` on the adapter config to opt into fetching. URL: ${url}`);
48
+ const response = await fetch(url);
49
+ if (!response.ok) throw new Error(`Failed to fetch image input (${response.status} ${response.statusText}): ${url}`);
50
+ const blob = await response.blob();
51
+ return {
52
+ imageBytes: arrayBufferToBase64(await blob.arrayBuffer()),
53
+ mimeType: part.source.mimeType || blob.type || "image/png"
54
+ };
56
55
  }
56
+ /**
57
+ * Convert an image or video prompt part into an Interactions API content
58
+ * block. Data sources become inline base64 `data`; URL sources pass through
59
+ * as `uri` (Files API URIs — mirrors the Interactions text adapter).
60
+ */
57
61
  function mediaPartToInteractionsContent(part) {
58
- const mimeType = part.source.mimeType;
59
- if (part.type === "image") {
60
- return part.source.type === "data" ? { type: "image", data: part.source.value, mime_type: mimeType } : { type: "image", uri: part.source.value, mime_type: mimeType };
61
- }
62
- return part.source.type === "data" ? { type: "video", data: part.source.value, mime_type: mimeType } : { type: "video", uri: part.source.value, mime_type: mimeType };
62
+ const mimeType = part.source.mimeType;
63
+ if (part.type === "image") return part.source.type === "data" ? {
64
+ type: "image",
65
+ data: part.source.value,
66
+ mime_type: mimeType
67
+ } : {
68
+ type: "image",
69
+ uri: part.source.value,
70
+ mime_type: mimeType
71
+ };
72
+ return part.source.type === "data" ? {
73
+ type: "video",
74
+ data: part.source.value,
75
+ mime_type: mimeType
76
+ } : {
77
+ type: "video",
78
+ uri: part.source.value,
79
+ mime_type: mimeType
80
+ };
63
81
  }
82
+ /**
83
+ * Pull the generated video out of a completed interaction. Prefers the
84
+ * SDK's `output_video` sugar, then walks `steps` back-to-front for the last
85
+ * `model_output` step carrying a video content block (the wire shape the
86
+ * raw REST response uses).
87
+ */
64
88
  function extractInteractionVideo(interaction) {
65
- const direct = interaction.output_video;
66
- if (direct && (direct.data || direct.uri)) {
67
- return {
68
- data: direct.data,
69
- uri: direct.uri,
70
- mimeType: direct.mime_type || "video/mp4"
71
- };
72
- }
73
- const steps = interaction.steps ?? [];
74
- for (let i = steps.length - 1; i >= 0; i--) {
75
- const step = steps[i];
76
- if (step?.type !== "model_output") continue;
77
- for (const block of step.content ?? []) {
78
- if (block.type === "video" && (block.data || block.uri)) {
79
- return {
80
- data: block.data,
81
- uri: block.uri,
82
- mimeType: block.mime_type || "video/mp4"
83
- };
84
- }
85
- }
86
- }
87
- return void 0;
89
+ const direct = interaction.output_video;
90
+ if (direct && (direct.data || direct.uri)) return {
91
+ data: direct.data,
92
+ uri: direct.uri,
93
+ mimeType: direct.mime_type || "video/mp4"
94
+ };
95
+ const steps = interaction.steps ?? [];
96
+ for (let i = steps.length - 1; i >= 0; i--) {
97
+ const step = steps[i];
98
+ if (step?.type !== "model_output") continue;
99
+ for (const block of step.content ?? []) if (block.type === "video" && (block.data || block.uri)) return {
100
+ data: block.data,
101
+ uri: block.uri,
102
+ mimeType: block.mime_type || "video/mp4"
103
+ };
104
+ }
88
105
  }
106
+ /**
107
+ * Map Interactions usage onto the canonical TokenUsage shape. Omni reports
108
+ * video output via `output_tokens_by_modality`; fall back to the video
109
+ * modality entry when the total is absent.
110
+ */
89
111
  function interactionUsageToTokenUsage(usage) {
90
- if (!usage) return void 0;
91
- const videoTokens = usage.output_tokens_by_modality?.find(
92
- (entry) => entry.modality === "video"
93
- )?.tokens;
94
- const promptTokens = usage.total_input_tokens ?? 0;
95
- const completionTokens = usage.total_output_tokens ?? videoTokens ?? 0;
96
- return {
97
- promptTokens,
98
- completionTokens,
99
- totalTokens: usage.total_tokens ?? promptTokens + completionTokens
100
- };
101
- }
102
- class GeminiVideoAdapter extends BaseVideoAdapter {
103
- name = "gemini";
104
- client;
105
- allowUrlFetch;
106
- constructor(config, model) {
107
- super({}, model);
108
- this.client = createGeminiClient(config);
109
- this.allowUrlFetch = config.allowUrlFetch ?? false;
110
- }
111
- async createVideoJob(options) {
112
- const { prompt, size, duration, logger } = options;
113
- logger.request(
114
- `activity=video.create provider=${this.name} model=${this.model} size=${size ?? "default"} duration=${duration ?? "default"}`,
115
- { provider: this.name, model: this.model }
116
- );
117
- if (isInteractionsVideoModel(this.model)) {
118
- return await this.createInteractionsVideoJob(options);
119
- }
120
- const modelOptions = options.modelOptions;
121
- try {
122
- const resolved = resolveMediaPrompt(prompt);
123
- if (resolved.videos.length > 0) {
124
- throw new Error(
125
- `${this.name}.createVideoJob does not support video prompt parts (model: ${this.model}).`
126
- );
127
- }
128
- if (resolved.audios.length > 0) {
129
- throw new Error(
130
- `${this.name}.createVideoJob does not support audio prompt parts (model: ${this.model}).`
131
- );
132
- }
133
- const { image, lastFrame, referenceImages } = await this.routeImageParts(
134
- resolved.images
135
- );
136
- const config = {
137
- ...modelOptions,
138
- ...size !== void 0 && { aspectRatio: size },
139
- ...duration !== void 0 && { durationSeconds: duration },
140
- ...lastFrame && { lastFrame },
141
- ...referenceImages.length > 0 && { referenceImages }
142
- };
143
- const operation = await this.client.models.generateVideos({
144
- model: this.model,
145
- prompt: resolved.text,
146
- ...image && { image },
147
- config
148
- });
149
- if (!operation.name) {
150
- throw new Error(
151
- "Veo did not return an operation name for the video generation job."
152
- );
153
- }
154
- return { jobId: operation.name, model: this.model };
155
- } catch (error) {
156
- logger.errors(`${this.name}.createVideoJob fatal`, {
157
- error,
158
- source: `${this.name}.createVideoJob`
159
- });
160
- throw error;
161
- }
162
- }
163
- /**
164
- * Gemini Omni Flash job creation via the Interactions API. Creates a
165
- * background interaction requesting video output; the interaction id is
166
- * the job id polled by `getVideoStatus` / `getVideoUrl`.
167
- */
168
- async createInteractionsVideoJob(options) {
169
- const { prompt, size, duration, logger } = options;
170
- const modelOptions = options.modelOptions;
171
- try {
172
- const resolved = resolveMediaPrompt(prompt);
173
- if (resolved.audios.length > 0) {
174
- throw new Error(
175
- `${this.name}.createVideoJob does not support audio prompt parts (model: ${this.model}).`
176
- );
177
- }
178
- const content = [
179
- ...resolved.images.map(mediaPartToInteractionsContent),
180
- ...resolved.videos.map(mediaPartToInteractionsContent)
181
- ];
182
- if (resolved.text) {
183
- content.push({ type: "text", text: resolved.text });
184
- }
185
- if (content.length === 0) {
186
- throw new Error(
187
- `${this.name}.createVideoJob: the prompt produced no content to send (model: ${this.model}).`
188
- );
189
- }
190
- const durations = this.availableDurations();
191
- if (duration !== void 0 && durations.kind === "range" && (duration < durations.min || duration > durations.max)) {
192
- throw new Error(
193
- `${this.name}.createVideoJob: duration ${duration}s is outside the ${durations.min}–${durations.max}s range supported by ${this.model}. Use snapDuration() to snap arbitrary values into range.`
194
- );
195
- }
196
- const responseFormat = size !== void 0 || duration !== void 0 ? {
197
- response_format: {
198
- type: "video",
199
- ...size !== void 0 && { aspect_ratio: size },
200
- ...duration !== void 0 && { duration: `${duration}s` }
201
- }
202
- } : {};
203
- const interaction = await this.client.interactions.create({
204
- ...modelOptions,
205
- model: this.model,
206
- input: [{ type: "user_input", content }],
207
- response_modalities: ["video"],
208
- background: true,
209
- ...responseFormat
210
- });
211
- if (!interaction.id) {
212
- throw new Error(
213
- "Gemini Omni did not return an interaction id for the video generation job."
214
- );
215
- }
216
- return { jobId: interaction.id, model: this.model };
217
- } catch (error) {
218
- logger.errors(`${this.name}.createVideoJob fatal`, {
219
- error,
220
- source: `${this.name}.createVideoJob`
221
- });
222
- throw error;
223
- }
224
- }
225
- /**
226
- * Route image prompt parts onto Veo's request fields by `metadata.role`.
227
- */
228
- async routeImageParts(parts) {
229
- let image;
230
- let lastFrame;
231
- const referenceImages = [];
232
- for (const part of parts) {
233
- const role = part.metadata?.role;
234
- switch (role) {
235
- case "end_frame": {
236
- if (lastFrame) {
237
- throw new Error(
238
- `${this.name}: Veo accepts at most one 'end_frame' image.`
239
- );
240
- }
241
- lastFrame = await imagePartToVeoImage(part, this.allowUrlFetch);
242
- break;
243
- }
244
- case "reference":
245
- case "character": {
246
- referenceImages.push({
247
- image: await imagePartToVeoImage(part, this.allowUrlFetch),
248
- referenceType: VideoGenerationReferenceType.ASSET
249
- });
250
- break;
251
- }
252
- case "start_frame":
253
- case void 0: {
254
- if (image) {
255
- throw new Error(
256
- `${this.name}: Veo accepts at most one starting image; received multiple 'start_frame'/un-roled images. Use metadata.role ('end_frame', 'reference') to disambiguate the others.`
257
- );
258
- }
259
- image = await imagePartToVeoImage(part, this.allowUrlFetch);
260
- break;
261
- }
262
- case "mask":
263
- case "control":
264
- throw new Error(
265
- `${this.name}: unsupported image role "${role}" for Veo video generation.`
266
- );
267
- }
268
- }
269
- return { image, lastFrame, referenceImages };
270
- }
271
- async getVideoStatus(jobId) {
272
- if (isInteractionsVideoModel(this.model)) {
273
- return await this.getInteractionsVideoStatus(jobId);
274
- }
275
- const operation = await this.getOperation(jobId);
276
- if (!operation.done) {
277
- return { jobId, status: "processing" };
278
- }
279
- if (operation.error) {
280
- return {
281
- jobId,
282
- status: "failed",
283
- error: operationErrorMessage(operation.error)
284
- };
285
- }
286
- const videos = operation.response?.generatedVideos ?? [];
287
- if (videos.length === 0) {
288
- const reasons = operation.response?.raiMediaFilteredReasons;
289
- return {
290
- jobId,
291
- status: "failed",
292
- error: reasons?.length ? `Video was filtered by Responsible-AI: ${reasons.join("; ")}` : "Veo returned no generated videos."
293
- };
294
- }
295
- return { jobId, status: "completed" };
296
- }
297
- /**
298
- * Poll an Omni background interaction. `in_progress` maps to
299
- * 'processing'; a `completed` interaction with no video content (e.g.
300
- * filtered output) is surfaced as a failure so `getVideoUrl` doesn't
301
- * throw on an empty response. `requires_action` also fails: the adapter
302
- * never sends tools, so it can only arise via
303
- * `previous_interaction_id` chaining onto a tool-bearing interaction —
304
- * and such an interaction never progresses without a client response,
305
- * so polling it would spin until timeout.
306
- */
307
- async getInteractionsVideoStatus(jobId) {
308
- const interaction = await this.getInteraction(jobId);
309
- const status = interaction.status;
310
- if (status === "in_progress") {
311
- return { jobId, status: "processing" };
312
- }
313
- if (status === "requires_action") {
314
- return {
315
- jobId,
316
- status: "failed",
317
- error: "Gemini Omni interaction is waiting on a client action (tool response), which the video jobs flow does not support."
318
- };
319
- }
320
- if (status === "completed") {
321
- if (!extractInteractionVideo(interaction)) {
322
- return {
323
- jobId,
324
- status: "failed",
325
- error: "Gemini Omni completed the interaction without returning a video (the output may have been filtered)."
326
- };
327
- }
328
- return { jobId, status: "completed" };
329
- }
330
- return {
331
- jobId,
332
- status: "failed",
333
- error: `Gemini Omni video generation ended with status "${status}".`
334
- };
335
- }
336
- async getVideoUrl(jobId) {
337
- if (isInteractionsVideoModel(this.model)) {
338
- return await this.getInteractionsVideoUrl(jobId);
339
- }
340
- const operation = await this.getOperation(jobId);
341
- if (!operation.done) {
342
- throw new Error(
343
- `Video is not ready yet. Check status first. Job ID: ${jobId}`
344
- );
345
- }
346
- if (operation.error) {
347
- throw new Error(
348
- `Video generation failed: ${operationErrorMessage(operation.error)}`
349
- );
350
- }
351
- const uri = operation.response?.generatedVideos?.[0]?.video?.uri;
352
- if (!uri) {
353
- const reasons = operation.response?.raiMediaFilteredReasons;
354
- throw new Error(
355
- reasons?.length ? `Video was filtered by Responsible-AI: ${reasons.join("; ")}` : `Video URL not found in operation response. Job ID: ${jobId}`
356
- );
357
- }
358
- return { jobId, url: uri };
359
- }
360
- /**
361
- * Extract the finished Omni video. Inline base64 output (the API default)
362
- * becomes a `data:` URL — matching the OpenAI Sora adapter's inline
363
- * delivery — and URI delivery passes through (Files API URIs need the API
364
- * key to download, like Veo). Usage carries the video-modality output
365
- * tokens (Omni bills per second of video, reported as tokens).
366
- */
367
- async getInteractionsVideoUrl(jobId) {
368
- const interaction = await this.getInteraction(jobId);
369
- const status = interaction.status;
370
- if (status === "in_progress") {
371
- throw new Error(
372
- `Video is not ready yet. Check status first. Job ID: ${jobId}`
373
- );
374
- }
375
- if (status !== "completed") {
376
- throw new Error(
377
- `Video generation failed: Gemini Omni interaction ended with status "${status}". Job ID: ${jobId}`
378
- );
379
- }
380
- const video = extractInteractionVideo(interaction);
381
- if (!video) {
382
- throw new Error(
383
- `Video not found in interaction response (the output may have been filtered). Job ID: ${jobId}`
384
- );
385
- }
386
- const usage = interactionUsageToTokenUsage(interaction.usage);
387
- const url = video.uri ?? `data:${video.mimeType};base64,${video.data}`;
388
- return { jobId, url, ...usage && { usage } };
389
- }
390
- availableDurations() {
391
- return getGeminiVideoDurationOptions(this.model);
392
- }
393
- snapDuration(seconds) {
394
- return snapToDurationOption(seconds, this.availableDurations());
395
- }
396
- /**
397
- * Fetch the long-running operation by name. The SDK's
398
- * `operations.getVideosOperation` needs a real `GenerateVideosOperation`
399
- * instance (it calls `_fromAPIResponse` on it), so reconstruct one from
400
- * the job ID rather than passing an object literal.
401
- */
402
- async getOperation(jobId) {
403
- const operation = new GenerateVideosOperation();
404
- operation.name = jobId;
405
- return await this.client.operations.getVideosOperation({ operation });
406
- }
407
- /**
408
- * Fetch an Omni background interaction by id.
409
- */
410
- async getInteraction(jobId) {
411
- return await this.client.interactions.get(jobId);
412
- }
112
+ if (!usage) return void 0;
113
+ const videoTokens = usage.output_tokens_by_modality?.find((entry) => entry.modality === "video")?.tokens;
114
+ const promptTokens = usage.total_input_tokens ?? 0;
115
+ const completionTokens = usage.total_output_tokens ?? videoTokens ?? 0;
116
+ return {
117
+ promptTokens,
118
+ completionTokens,
119
+ totalTokens: usage.total_tokens ?? promptTokens + completionTokens
120
+ };
413
121
  }
122
+ /**
123
+ * Gemini Video Generation Adapter (Veo + Gemini Omni Flash)
124
+ *
125
+ * Tree-shakeable adapter for Google video generation, routing by model:
126
+ *
127
+ * **Veo models** run as a long-running operation: `createVideoJob` starts
128
+ * the operation via the `:predictLongRunning` endpoint, `getVideoStatus`
129
+ * polls it, and `getVideoUrl` extracts the generated video's URI once it
130
+ * completes. Image prompt parts are routed by `metadata.role`:
131
+ * - `'start_frame'` (or the first un-roled image) → the input image the
132
+ * video starts from
133
+ * - `'end_frame'` → `lastFrame` (the frame the video ends on)
134
+ * - `'reference'` / `'character'` → `referenceImages` (asset references,
135
+ * Veo 3.1)
136
+ *
137
+ * Note: the returned Veo video URI is served by the Gemini Files API and
138
+ * requires the API key (`x-goog-api-key` header or `?key=` query
139
+ * parameter) to download.
140
+ *
141
+ * **Gemini Omni Flash** (`gemini-omni-flash-preview`) only serves the
142
+ * Interactions API: `createVideoJob` creates a background interaction with
143
+ * `response_modalities: ['video']`, `getVideoStatus` polls it by id, and
144
+ * `getVideoUrl` returns the inline base64 MP4 as a `data:` URL (or the
145
+ * Files API URI when the server delivers by reference). Image and video
146
+ * prompt parts are sent as interaction content blocks, grouped as images,
147
+ * then videos, then the text prompt (interleaving is not preserved); pass
148
+ * `modelOptions.previous_interaction_id` to conversationally edit a prior
149
+ * Omni generation.
150
+ *
151
+ * @experimental Video generation is an experimental feature and may change.
152
+ */
153
+ var GeminiVideoAdapter = class extends BaseVideoAdapter {
154
+ name = "gemini";
155
+ client;
156
+ allowUrlFetch;
157
+ constructor(config, model) {
158
+ super({}, model);
159
+ this.client = createGeminiClient(config);
160
+ this.allowUrlFetch = config.allowUrlFetch ?? false;
161
+ }
162
+ async createVideoJob(options) {
163
+ const { prompt, size, duration, logger } = options;
164
+ logger.request(`activity=video.create provider=${this.name} model=${this.model} size=${size ?? "default"} duration=${duration ?? "default"}`, {
165
+ provider: this.name,
166
+ model: this.model
167
+ });
168
+ if (isInteractionsVideoModel(this.model)) return await this.createInteractionsVideoJob(options);
169
+ const modelOptions = options.modelOptions;
170
+ try {
171
+ const resolved = resolveMediaPrompt(prompt);
172
+ if (resolved.videos.length > 0) throw new Error(`${this.name}.createVideoJob does not support video prompt parts (model: ${this.model}).`);
173
+ if (resolved.audios.length > 0) throw new Error(`${this.name}.createVideoJob does not support audio prompt parts (model: ${this.model}).`);
174
+ const { image, lastFrame, referenceImages } = await this.routeImageParts(resolved.images);
175
+ const config = {
176
+ ...modelOptions,
177
+ ...size !== void 0 && { aspectRatio: size },
178
+ ...duration !== void 0 && { durationSeconds: duration },
179
+ ...lastFrame && { lastFrame },
180
+ ...referenceImages.length > 0 && { referenceImages }
181
+ };
182
+ const operation = await this.client.models.generateVideos({
183
+ model: this.model,
184
+ prompt: resolved.text,
185
+ ...image && { image },
186
+ config
187
+ });
188
+ if (!operation.name) throw new Error("Veo did not return an operation name for the video generation job.");
189
+ return {
190
+ jobId: operation.name,
191
+ model: this.model
192
+ };
193
+ } catch (error) {
194
+ logger.errors(`${this.name}.createVideoJob fatal`, {
195
+ error,
196
+ source: `${this.name}.createVideoJob`
197
+ });
198
+ throw error;
199
+ }
200
+ }
201
+ /**
202
+ * Gemini Omni Flash job creation via the Interactions API. Creates a
203
+ * background interaction requesting video output; the interaction id is
204
+ * the job id polled by `getVideoStatus` / `getVideoUrl`.
205
+ */
206
+ async createInteractionsVideoJob(options) {
207
+ const { prompt, size, duration, logger } = options;
208
+ const modelOptions = options.modelOptions;
209
+ try {
210
+ const resolved = resolveMediaPrompt(prompt);
211
+ if (resolved.audios.length > 0) throw new Error(`${this.name}.createVideoJob does not support audio prompt parts (model: ${this.model}).`);
212
+ const content = [...resolved.images.map(mediaPartToInteractionsContent), ...resolved.videos.map(mediaPartToInteractionsContent)];
213
+ if (resolved.text) content.push({
214
+ type: "text",
215
+ text: resolved.text
216
+ });
217
+ if (content.length === 0) throw new Error(`${this.name}.createVideoJob: the prompt produced no content to send (model: ${this.model}).`);
218
+ const durations = this.availableDurations();
219
+ if (duration !== void 0 && durations.kind === "range" && (duration < durations.min || duration > durations.max)) throw new Error(`${this.name}.createVideoJob: duration ${duration}s is outside the ${durations.min}–${durations.max}s range supported by ${this.model}. Use snapDuration() to snap arbitrary values into range.`);
220
+ const responseFormat = size !== void 0 || duration !== void 0 ? { response_format: {
221
+ type: "video",
222
+ ...size !== void 0 && { aspect_ratio: size },
223
+ ...duration !== void 0 && { duration: `${duration}s` }
224
+ } } : {};
225
+ const interaction = await this.client.interactions.create({
226
+ ...modelOptions,
227
+ model: this.model,
228
+ input: [{
229
+ type: "user_input",
230
+ content
231
+ }],
232
+ response_modalities: ["video"],
233
+ background: true,
234
+ ...responseFormat
235
+ });
236
+ if (!interaction.id) throw new Error("Gemini Omni did not return an interaction id for the video generation job.");
237
+ return {
238
+ jobId: interaction.id,
239
+ model: this.model
240
+ };
241
+ } catch (error) {
242
+ logger.errors(`${this.name}.createVideoJob fatal`, {
243
+ error,
244
+ source: `${this.name}.createVideoJob`
245
+ });
246
+ throw error;
247
+ }
248
+ }
249
+ /**
250
+ * Route image prompt parts onto Veo's request fields by `metadata.role`.
251
+ */
252
+ async routeImageParts(parts) {
253
+ let image;
254
+ let lastFrame;
255
+ const referenceImages = [];
256
+ for (const part of parts) {
257
+ const role = part.metadata?.role;
258
+ switch (role) {
259
+ case "end_frame":
260
+ if (lastFrame) throw new Error(`${this.name}: Veo accepts at most one 'end_frame' image.`);
261
+ lastFrame = await imagePartToVeoImage(part, this.allowUrlFetch);
262
+ break;
263
+ case "reference":
264
+ case "character":
265
+ referenceImages.push({
266
+ image: await imagePartToVeoImage(part, this.allowUrlFetch),
267
+ referenceType: VideoGenerationReferenceType.ASSET
268
+ });
269
+ break;
270
+ case "start_frame":
271
+ case void 0:
272
+ if (image) throw new Error(`${this.name}: Veo accepts at most one starting image; received multiple 'start_frame'/un-roled images. Use metadata.role ('end_frame', 'reference') to disambiguate the others.`);
273
+ image = await imagePartToVeoImage(part, this.allowUrlFetch);
274
+ break;
275
+ case "mask":
276
+ case "control": throw new Error(`${this.name}: unsupported image role "${role}" for Veo video generation.`);
277
+ }
278
+ }
279
+ return {
280
+ image,
281
+ lastFrame,
282
+ referenceImages
283
+ };
284
+ }
285
+ async getVideoStatus(jobId) {
286
+ if (isInteractionsVideoModel(this.model)) return await this.getInteractionsVideoStatus(jobId);
287
+ const operation = await this.getOperation(jobId);
288
+ if (!operation.done) return {
289
+ jobId,
290
+ status: "processing"
291
+ };
292
+ if (operation.error) return {
293
+ jobId,
294
+ status: "failed",
295
+ error: operationErrorMessage(operation.error)
296
+ };
297
+ if ((operation.response?.generatedVideos ?? []).length === 0) {
298
+ const reasons = operation.response?.raiMediaFilteredReasons;
299
+ return {
300
+ jobId,
301
+ status: "failed",
302
+ error: reasons?.length ? `Video was filtered by Responsible-AI: ${reasons.join("; ")}` : "Veo returned no generated videos."
303
+ };
304
+ }
305
+ return {
306
+ jobId,
307
+ status: "completed"
308
+ };
309
+ }
310
+ /**
311
+ * Poll an Omni background interaction. `in_progress` maps to
312
+ * 'processing'; a `completed` interaction with no video content (e.g.
313
+ * filtered output) is surfaced as a failure so `getVideoUrl` doesn't
314
+ * throw on an empty response. `requires_action` also fails: the adapter
315
+ * never sends tools, so it can only arise via
316
+ * `previous_interaction_id` chaining onto a tool-bearing interaction —
317
+ * and such an interaction never progresses without a client response,
318
+ * so polling it would spin until timeout.
319
+ */
320
+ async getInteractionsVideoStatus(jobId) {
321
+ const interaction = await this.getInteraction(jobId);
322
+ const status = interaction.status;
323
+ if (status === "in_progress") return {
324
+ jobId,
325
+ status: "processing"
326
+ };
327
+ if (status === "requires_action") return {
328
+ jobId,
329
+ status: "failed",
330
+ error: "Gemini Omni interaction is waiting on a client action (tool response), which the video jobs flow does not support."
331
+ };
332
+ if (status === "completed") {
333
+ if (!extractInteractionVideo(interaction)) return {
334
+ jobId,
335
+ status: "failed",
336
+ error: "Gemini Omni completed the interaction without returning a video (the output may have been filtered)."
337
+ };
338
+ return {
339
+ jobId,
340
+ status: "completed"
341
+ };
342
+ }
343
+ return {
344
+ jobId,
345
+ status: "failed",
346
+ error: `Gemini Omni video generation ended with status "${status}".`
347
+ };
348
+ }
349
+ async getVideoUrl(jobId) {
350
+ if (isInteractionsVideoModel(this.model)) return await this.getInteractionsVideoUrl(jobId);
351
+ const operation = await this.getOperation(jobId);
352
+ if (!operation.done) throw new Error(`Video is not ready yet. Check status first. Job ID: ${jobId}`);
353
+ if (operation.error) throw new Error(`Video generation failed: ${operationErrorMessage(operation.error)}`);
354
+ const uri = operation.response?.generatedVideos?.[0]?.video?.uri;
355
+ if (!uri) {
356
+ const reasons = operation.response?.raiMediaFilteredReasons;
357
+ throw new Error(reasons?.length ? `Video was filtered by Responsible-AI: ${reasons.join("; ")}` : `Video URL not found in operation response. Job ID: ${jobId}`);
358
+ }
359
+ return {
360
+ jobId,
361
+ url: uri
362
+ };
363
+ }
364
+ /**
365
+ * Extract the finished Omni video. Inline base64 output (the API default)
366
+ * becomes a `data:` URL — matching the OpenAI Sora adapter's inline
367
+ * delivery — and URI delivery passes through (Files API URIs need the API
368
+ * key to download, like Veo). Usage carries the video-modality output
369
+ * tokens (Omni bills per second of video, reported as tokens).
370
+ */
371
+ async getInteractionsVideoUrl(jobId) {
372
+ const interaction = await this.getInteraction(jobId);
373
+ const status = interaction.status;
374
+ if (status === "in_progress") throw new Error(`Video is not ready yet. Check status first. Job ID: ${jobId}`);
375
+ if (status !== "completed") throw new Error(`Video generation failed: Gemini Omni interaction ended with status "${status}". Job ID: ${jobId}`);
376
+ const video = extractInteractionVideo(interaction);
377
+ if (!video) throw new Error(`Video not found in interaction response (the output may have been filtered). Job ID: ${jobId}`);
378
+ const usage = interactionUsageToTokenUsage(interaction.usage);
379
+ return {
380
+ jobId,
381
+ url: video.uri ?? `data:${video.mimeType};base64,${video.data}`,
382
+ ...usage && { usage }
383
+ };
384
+ }
385
+ availableDurations() {
386
+ return getGeminiVideoDurationOptions(this.model);
387
+ }
388
+ snapDuration(seconds) {
389
+ return snapToDurationOption(seconds, this.availableDurations());
390
+ }
391
+ /**
392
+ * Fetch the long-running operation by name. The SDK's
393
+ * `operations.getVideosOperation` needs a real `GenerateVideosOperation`
394
+ * instance (it calls `_fromAPIResponse` on it), so reconstruct one from
395
+ * the job ID rather than passing an object literal.
396
+ */
397
+ async getOperation(jobId) {
398
+ const operation = new GenerateVideosOperation();
399
+ operation.name = jobId;
400
+ return await this.client.operations.getVideosOperation({ operation });
401
+ }
402
+ /**
403
+ * Fetch an Omni background interaction by id.
404
+ */
405
+ async getInteraction(jobId) {
406
+ return await this.client.interactions.get(jobId);
407
+ }
408
+ };
409
+ /**
410
+ * Creates a Gemini video adapter with an explicit API key.
411
+ * Type resolution happens here at the call site.
412
+ *
413
+ * @experimental Video generation is an experimental feature and may change.
414
+ *
415
+ * @param model - The model name (e.g., 'veo-3.1-generate-preview')
416
+ * @param apiKey - Your Google API key
417
+ * @param config - Optional additional configuration
418
+ * @returns Configured Gemini video adapter instance with resolved types
419
+ *
420
+ * @example
421
+ * ```typescript
422
+ * const adapter = createGeminiVideo('veo-3.1-generate-preview', 'your-api-key');
423
+ *
424
+ * const { jobId } = await generateVideo({
425
+ * adapter,
426
+ * prompt: 'A beautiful sunset over the ocean',
427
+ * duration: adapter.snapDuration(7), // → 6
428
+ * });
429
+ * ```
430
+ */
414
431
  function createGeminiVideo(model, apiKey, config) {
415
- return new GeminiVideoAdapter({ apiKey, ...config }, model);
432
+ return new GeminiVideoAdapter({
433
+ apiKey,
434
+ ...config
435
+ }, model);
416
436
  }
437
+ /**
438
+ * Creates a Gemini video adapter with automatic API key detection from environment variables.
439
+ * Type resolution happens here at the call site.
440
+ *
441
+ * Looks for `GOOGLE_API_KEY` or `GEMINI_API_KEY` in:
442
+ * - `process.env` (Node.js)
443
+ * - `window.env` (Browser with injected env)
444
+ *
445
+ * @experimental Video generation is an experimental feature and may change.
446
+ *
447
+ * @param model - The model name (e.g., 'veo-3.1-generate-preview')
448
+ * @param config - Optional configuration (excluding apiKey which is auto-detected)
449
+ * @returns Configured Gemini video adapter instance with resolved types
450
+ * @throws Error if GOOGLE_API_KEY or GEMINI_API_KEY is not found in environment
451
+ *
452
+ * @example
453
+ * ```typescript
454
+ * // Automatically uses GOOGLE_API_KEY from environment
455
+ * const adapter = geminiVideo('veo-3.1-generate-preview');
456
+ *
457
+ * // Create a video generation job
458
+ * const { jobId } = await generateVideo({
459
+ * adapter,
460
+ * prompt: 'A cat playing piano'
461
+ * });
462
+ *
463
+ * // Poll for status
464
+ * const status = await getVideoJobStatus({ adapter, jobId });
465
+ * ```
466
+ */
417
467
  function geminiVideo(model, config) {
418
- const apiKey = getGeminiApiKeyFromEnv();
419
- return createGeminiVideo(model, apiKey, config);
468
+ return createGeminiVideo(model, getGeminiApiKeyFromEnv(), config);
420
469
  }
421
- export {
422
- GeminiVideoAdapter,
423
- createGeminiVideo,
424
- geminiVideo
425
- };
426
- //# sourceMappingURL=video.js.map
470
+ //#endregion
471
+ export { GeminiVideoAdapter, createGeminiVideo, geminiVideo };
472
+
473
+ //# sourceMappingURL=video.js.map