@tanstack/ai-gemini 0.26.4 → 0.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -15
- package/dist/esm/adapters/text.d.ts +9 -0
- package/dist/esm/adapters/text.js +165 -13
- package/dist/esm/adapters/text.js.map +1 -1
- package/dist/esm/adapters/video.d.ts +13 -5
- package/dist/esm/adapters/video.js +14 -58
- package/dist/esm/adapters/video.js.map +1 -1
- package/dist/esm/files/index.d.ts +51 -0
- package/dist/esm/files/index.js +59 -0
- package/dist/esm/files/index.js.map +1 -0
- package/dist/esm/index.d.ts +4 -3
- package/dist/esm/index.js +3 -2
- package/dist/esm/message-types.d.ts +36 -0
- package/dist/esm/model-meta.d.ts +7 -6
- package/dist/esm/model-meta.js +39 -7
- package/dist/esm/model-meta.js.map +1 -1
- package/dist/esm/tools/tool-converter.js +1 -1
- package/dist/esm/tools/tool-converter.js.map +1 -1
- package/dist/esm/video/video-provider-options.d.ts +32 -7
- package/dist/esm/video/video-provider-options.js +20 -1
- package/dist/esm/video/video-provider-options.js.map +1 -1
- package/package.json +3 -3
- package/src/adapters/text.ts +253 -20
- package/src/adapters/video.ts +43 -12
- package/src/files/index.ts +114 -0
- package/src/index.ts +12 -0
- package/src/message-types.ts +37 -0
- package/src/model-meta.ts +47 -6
- package/src/tools/tool-converter.ts +3 -7
- package/src/video/video-provider-options.ts +52 -7
package/src/adapters/text.ts
CHANGED
|
@@ -28,6 +28,7 @@ import type {
|
|
|
28
28
|
GoogleGenAI,
|
|
29
29
|
Part,
|
|
30
30
|
ThinkingLevel,
|
|
31
|
+
VideoMetadata,
|
|
31
32
|
} from '@google/genai'
|
|
32
33
|
import type {
|
|
33
34
|
ContentPart,
|
|
@@ -40,9 +41,113 @@ import type { ExternalTextProviderOptions } from '../text/text-provider-options'
|
|
|
40
41
|
import type {
|
|
41
42
|
GeminiMessageMetadataByModality,
|
|
42
43
|
GeminiToolCallMetadata,
|
|
44
|
+
GeminiVideoMetadata,
|
|
45
|
+
GeminiVideoProcessing,
|
|
43
46
|
} from '../message-types'
|
|
44
47
|
import type { GeminiClientConfig } from '../utils/client'
|
|
45
48
|
|
|
49
|
+
/**
|
|
50
|
+
* Fallback MIME types for URL-sourced media parts that don't specify one.
|
|
51
|
+
*/
|
|
52
|
+
const DEFAULT_MEDIA_MIME_TYPES = {
|
|
53
|
+
image: 'image/jpeg',
|
|
54
|
+
audio: 'audio/mp3',
|
|
55
|
+
video: 'video/mp4',
|
|
56
|
+
document: 'application/pdf',
|
|
57
|
+
} as const
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* Content block shape for an Interactions API `input` step. The installed
|
|
61
|
+
* @google/genai types predate the video `processing` field, so we model the
|
|
62
|
+
* subset we emit and cast at the call site.
|
|
63
|
+
*/
|
|
64
|
+
type InteractionContent =
|
|
65
|
+
| { type: 'text'; text: string }
|
|
66
|
+
| {
|
|
67
|
+
type: 'video'
|
|
68
|
+
uri?: string
|
|
69
|
+
data?: string
|
|
70
|
+
mime_type?: string
|
|
71
|
+
processing?: GeminiVideoProcessing
|
|
72
|
+
}
|
|
73
|
+
| {
|
|
74
|
+
type: 'image' | 'audio' | 'document'
|
|
75
|
+
uri?: string
|
|
76
|
+
data?: string
|
|
77
|
+
mime_type?: string
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
interface InteractionStep {
|
|
81
|
+
type: 'user_input' | 'model_output'
|
|
82
|
+
content: Array<InteractionContent>
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/** True when any message carries a video part requesting agentic processing. */
|
|
86
|
+
function hasAgenticVideo(messages: Array<ModelMessage>): boolean {
|
|
87
|
+
return messages.some(
|
|
88
|
+
(msg) =>
|
|
89
|
+
Array.isArray(msg.content) &&
|
|
90
|
+
msg.content.some(
|
|
91
|
+
(part) =>
|
|
92
|
+
part.type === 'video' &&
|
|
93
|
+
(part.metadata as GeminiVideoMetadata | undefined)?.processing ===
|
|
94
|
+
'agentic',
|
|
95
|
+
),
|
|
96
|
+
)
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/** Convert a single content part to an Interactions API content block. */
|
|
100
|
+
function contentPartToInteraction(part: ContentPart): InteractionContent {
|
|
101
|
+
if (part.type === 'text') {
|
|
102
|
+
return { type: 'text', text: part.content }
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
const source = part.source
|
|
106
|
+
const mimeType =
|
|
107
|
+
source.type === 'data'
|
|
108
|
+
? source.mimeType
|
|
109
|
+
: (source.mimeType ?? DEFAULT_MEDIA_MIME_TYPES[part.type])
|
|
110
|
+
const base =
|
|
111
|
+
source.type === 'data'
|
|
112
|
+
? { data: source.value, mime_type: mimeType }
|
|
113
|
+
: { uri: source.value, mime_type: mimeType }
|
|
114
|
+
|
|
115
|
+
if (part.type === 'video') {
|
|
116
|
+
const processing = (part.metadata as GeminiVideoMetadata | undefined)
|
|
117
|
+
?.processing
|
|
118
|
+
return { type: 'video', ...base, ...(processing && { processing }) }
|
|
119
|
+
}
|
|
120
|
+
return { type: part.type, ...base }
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Build the Interactions API `input` from chat messages. Each user/assistant
|
|
125
|
+
* message becomes a `user_input` / `model_output` step wrapping its content
|
|
126
|
+
* blocks — the wrapping the Python SDK performs implicitly but the JS SDK
|
|
127
|
+
* does not. Tool messages are skipped (unsupported on this path).
|
|
128
|
+
*/
|
|
129
|
+
function buildInteractionsInput(
|
|
130
|
+
messages: Array<ModelMessage>,
|
|
131
|
+
): Array<InteractionStep> {
|
|
132
|
+
const steps: Array<InteractionStep> = []
|
|
133
|
+
for (const msg of messages) {
|
|
134
|
+
if (msg.role === 'tool') continue
|
|
135
|
+
const stepType = msg.role === 'assistant' ? 'model_output' : 'user_input'
|
|
136
|
+
const content: Array<InteractionContent> = []
|
|
137
|
+
if (Array.isArray(msg.content)) {
|
|
138
|
+
for (const part of msg.content) {
|
|
139
|
+
content.push(contentPartToInteraction(part))
|
|
140
|
+
}
|
|
141
|
+
} else if (msg.content) {
|
|
142
|
+
content.push({ type: 'text', text: msg.content })
|
|
143
|
+
}
|
|
144
|
+
if (content.length > 0) {
|
|
145
|
+
steps.push({ type: stepType, content })
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
return steps
|
|
149
|
+
}
|
|
150
|
+
|
|
46
151
|
/**
|
|
47
152
|
* Configuration for Gemini text adapter
|
|
48
153
|
*/
|
|
@@ -122,6 +227,13 @@ export class GeminiTextAdapter<
|
|
|
122
227
|
async *chatStream(
|
|
123
228
|
options: TextOptions<GeminiTextProviderOptions>,
|
|
124
229
|
): AsyncIterable<AdapterYieldChunk> {
|
|
230
|
+
// Agentic video understanding is only exposed through the Interactions API,
|
|
231
|
+
// not generateContent. Detect it and take that path instead.
|
|
232
|
+
if (hasAgenticVideo(options.messages)) {
|
|
233
|
+
yield* this.interactionsStream(options)
|
|
234
|
+
return
|
|
235
|
+
}
|
|
236
|
+
|
|
125
237
|
const mappedOptions = this.mapCommonOptionsToGemini(options)
|
|
126
238
|
const { logger } = options
|
|
127
239
|
|
|
@@ -161,6 +273,114 @@ export class GeminiTextAdapter<
|
|
|
161
273
|
}
|
|
162
274
|
}
|
|
163
275
|
|
|
276
|
+
/**
|
|
277
|
+
* Agentic video-understanding path via the Interactions API.
|
|
278
|
+
*
|
|
279
|
+
* The Interactions API (unlike `generateContent`) requires message parts to
|
|
280
|
+
* be wrapped in `user_input` / `model_output` steps, and it accepts the
|
|
281
|
+
* `processing: 'agentic'` video flag. This is a non-streaming call whose
|
|
282
|
+
* single text result is re-emitted as AG-UI stream chunks.
|
|
283
|
+
*/
|
|
284
|
+
private async *interactionsStream(
|
|
285
|
+
options: TextOptions<GeminiTextProviderOptions>,
|
|
286
|
+
): AsyncIterable<AdapterYieldChunk> {
|
|
287
|
+
const model = options.model
|
|
288
|
+
const { logger } = options
|
|
289
|
+
const runId = options.runId ?? generateId(this.name)
|
|
290
|
+
const threadId = options.threadId ?? generateId(this.name)
|
|
291
|
+
const messageId = generateId(this.name)
|
|
292
|
+
|
|
293
|
+
try {
|
|
294
|
+
logger.request(
|
|
295
|
+
`activity=chat provider=gemini model=${model} messages=${options.messages.length} mode=interactions-agentic-video`,
|
|
296
|
+
{ provider: 'gemini', model },
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
const normalizedPrompts = normalizeSystemPrompts(options.systemPrompts)
|
|
300
|
+
const systemInstruction =
|
|
301
|
+
normalizedPrompts.length > 0
|
|
302
|
+
? normalizedPrompts.map((p) => p.content).join('\n')
|
|
303
|
+
: undefined
|
|
304
|
+
|
|
305
|
+
const input = buildInteractionsInput(options.messages)
|
|
306
|
+
|
|
307
|
+
// The installed @google/genai (2.10.0) Interactions `VideoContent` type
|
|
308
|
+
// predates the `processing` field, so the structurally-built input is
|
|
309
|
+
// cast at the call boundary. The SDK forwards it to the wire unchanged.
|
|
310
|
+
const interaction = await this.client.interactions.create({
|
|
311
|
+
model,
|
|
312
|
+
...(systemInstruction !== undefined && {
|
|
313
|
+
system_instruction: systemInstruction,
|
|
314
|
+
}),
|
|
315
|
+
input: input as never,
|
|
316
|
+
})
|
|
317
|
+
|
|
318
|
+
const text = interaction.output_text ?? ''
|
|
319
|
+
|
|
320
|
+
yield {
|
|
321
|
+
type: EventType.RUN_STARTED,
|
|
322
|
+
runId,
|
|
323
|
+
threadId,
|
|
324
|
+
model,
|
|
325
|
+
timestamp: Date.now(),
|
|
326
|
+
parentRunId: options.parentRunId,
|
|
327
|
+
}
|
|
328
|
+
yield {
|
|
329
|
+
type: EventType.TEXT_MESSAGE_START,
|
|
330
|
+
messageId,
|
|
331
|
+
model,
|
|
332
|
+
timestamp: Date.now(),
|
|
333
|
+
role: 'assistant',
|
|
334
|
+
}
|
|
335
|
+
if (text) {
|
|
336
|
+
yield {
|
|
337
|
+
type: EventType.TEXT_MESSAGE_CONTENT,
|
|
338
|
+
messageId,
|
|
339
|
+
model,
|
|
340
|
+
timestamp: Date.now(),
|
|
341
|
+
delta: text,
|
|
342
|
+
content: text,
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
yield {
|
|
346
|
+
type: EventType.TEXT_MESSAGE_END,
|
|
347
|
+
messageId,
|
|
348
|
+
model,
|
|
349
|
+
timestamp: Date.now(),
|
|
350
|
+
}
|
|
351
|
+
yield {
|
|
352
|
+
type: EventType.RUN_FINISHED,
|
|
353
|
+
runId,
|
|
354
|
+
threadId,
|
|
355
|
+
model,
|
|
356
|
+
timestamp: Date.now(),
|
|
357
|
+
finishReason: 'stop',
|
|
358
|
+
}
|
|
359
|
+
} catch (error) {
|
|
360
|
+
const rawEvent = toRunErrorRawEvent(error)
|
|
361
|
+
logger.errors('gemini.interactionsStream fatal', {
|
|
362
|
+
error,
|
|
363
|
+
source: 'gemini.interactionsStream',
|
|
364
|
+
})
|
|
365
|
+
yield {
|
|
366
|
+
type: EventType.RUN_ERROR,
|
|
367
|
+
model,
|
|
368
|
+
timestamp: Date.now(),
|
|
369
|
+
message:
|
|
370
|
+
error instanceof Error
|
|
371
|
+
? error.message
|
|
372
|
+
: 'An unknown error occurred during the chat stream.',
|
|
373
|
+
...(rawEvent !== undefined && { rawEvent }),
|
|
374
|
+
error: {
|
|
375
|
+
message:
|
|
376
|
+
error instanceof Error
|
|
377
|
+
? error.message
|
|
378
|
+
: 'An unknown error occurred during the chat stream.',
|
|
379
|
+
},
|
|
380
|
+
}
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
|
|
164
384
|
/**
|
|
165
385
|
* Generate structured output using Gemini's native JSON response format.
|
|
166
386
|
* Uses responseMimeType: 'application/json' and responseSchema for structured output.
|
|
@@ -595,29 +815,42 @@ export class GeminiTextAdapter<
|
|
|
595
815
|
case 'audio':
|
|
596
816
|
case 'video':
|
|
597
817
|
case 'document': {
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
818
|
+
const geminiPart: Part =
|
|
819
|
+
part.source.type === 'data'
|
|
820
|
+
? {
|
|
821
|
+
inlineData: {
|
|
822
|
+
data: part.source.value,
|
|
823
|
+
mimeType: part.source.mimeType,
|
|
824
|
+
},
|
|
825
|
+
}
|
|
826
|
+
: {
|
|
827
|
+
fileData: {
|
|
828
|
+
fileUri: part.source.value,
|
|
829
|
+
// For URL sources, use provided mimeType or fall back to
|
|
830
|
+
// reasonable defaults.
|
|
831
|
+
mimeType:
|
|
832
|
+
part.source.mimeType ?? DEFAULT_MEDIA_MIME_TYPES[part.type],
|
|
833
|
+
},
|
|
834
|
+
}
|
|
835
|
+
|
|
836
|
+
// Apply single-pass video sampling controls (fps / clip offsets) from
|
|
837
|
+
// the part metadata. `processing: 'agentic'` is handled separately via
|
|
838
|
+
// the Interactions API and never reaches this generateContent path.
|
|
839
|
+
if (part.type === 'video') {
|
|
840
|
+
const meta = part.metadata as GeminiVideoMetadata | undefined
|
|
841
|
+
const videoMetadata: VideoMetadata = {
|
|
842
|
+
...(meta?.fps !== undefined && { fps: meta.fps }),
|
|
843
|
+
...(meta?.startOffset !== undefined && {
|
|
844
|
+
startOffset: meta.startOffset,
|
|
845
|
+
}),
|
|
846
|
+
...(meta?.endOffset !== undefined && { endOffset: meta.endOffset }),
|
|
604
847
|
}
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
const defaultMimeType = {
|
|
608
|
-
image: 'image/jpeg',
|
|
609
|
-
audio: 'audio/mp3',
|
|
610
|
-
video: 'video/mp4',
|
|
611
|
-
document: 'application/pdf',
|
|
612
|
-
}[part.type]
|
|
613
|
-
|
|
614
|
-
return {
|
|
615
|
-
fileData: {
|
|
616
|
-
fileUri: part.source.value,
|
|
617
|
-
mimeType: part.source.mimeType ?? defaultMimeType,
|
|
618
|
-
},
|
|
848
|
+
if (Object.keys(videoMetadata).length > 0) {
|
|
849
|
+
geminiPart.videoMetadata = videoMetadata
|
|
619
850
|
}
|
|
620
851
|
}
|
|
852
|
+
|
|
853
|
+
return geminiPart
|
|
621
854
|
}
|
|
622
855
|
default: {
|
|
623
856
|
const _exhaustiveCheck: never = part
|
package/src/adapters/video.ts
CHANGED
|
@@ -9,6 +9,7 @@ import { createGeminiClient, getGeminiApiKeyFromEnv } from '../utils'
|
|
|
9
9
|
import {
|
|
10
10
|
getGeminiVideoDurationOptions,
|
|
11
11
|
isInteractionsVideoModel,
|
|
12
|
+
parseGeminiOmniVideoSize,
|
|
12
13
|
} from '../video/video-provider-options'
|
|
13
14
|
import type { DurationOptions } from '@tanstack/ai/adapters'
|
|
14
15
|
import type {
|
|
@@ -36,7 +37,6 @@ import type {
|
|
|
36
37
|
GeminiVideoModelProviderOptionsByName,
|
|
37
38
|
GeminiVideoModelSizeByName,
|
|
38
39
|
GeminiVideoProviderOptions,
|
|
39
|
-
GeminiVideoSize,
|
|
40
40
|
} from '../video/video-provider-options'
|
|
41
41
|
import type { GeminiClientConfig } from '../utils/client'
|
|
42
42
|
|
|
@@ -228,15 +228,19 @@ function interactionUsageToTokenUsage(
|
|
|
228
228
|
* requires the API key (`x-goog-api-key` header or `?key=` query
|
|
229
229
|
* parameter) to download.
|
|
230
230
|
*
|
|
231
|
-
* **Gemini Omni Flash** (`gemini-omni-flash
|
|
232
|
-
*
|
|
231
|
+
* **Gemini Omni Flash** (`gemini-omni-1.1-flash`, plus the deprecated
|
|
232
|
+
* `gemini-omni-flash-preview` alias) only serves the Interactions API:
|
|
233
|
+
* `createVideoJob` creates a background interaction with
|
|
233
234
|
* `response_modalities: ['video']`, `getVideoStatus` polls it by id, and
|
|
234
235
|
* `getVideoUrl` returns the inline base64 MP4 as a `data:` URL (or the
|
|
235
236
|
* Files API URI when the server delivers by reference). Image and video
|
|
236
237
|
* prompt parts are sent as interaction content blocks, grouped as images,
|
|
237
238
|
* then videos, then the text prompt (interleaving is not preserved); pass
|
|
238
239
|
* `modelOptions.previous_interaction_id` to conversationally edit a prior
|
|
239
|
-
* Omni generation.
|
|
240
|
+
* Omni generation. `size` is an `aspectRatio_resolution` template
|
|
241
|
+
* (`'16:9'` or `'16:9_1080p'`); the optional suffix maps onto
|
|
242
|
+
* `response_format.resolution` (`'360p' | '720p' | '1080p' | '4k'`,
|
|
243
|
+
* default 720p).
|
|
240
244
|
*
|
|
241
245
|
* @experimental Video generation is an experimental feature and may change.
|
|
242
246
|
*/
|
|
@@ -264,7 +268,7 @@ export class GeminiVideoAdapter<
|
|
|
264
268
|
async createVideoJob(
|
|
265
269
|
options: VideoGenerationOptions<
|
|
266
270
|
GeminiVideoModelProviderOptionsByName[TModel],
|
|
267
|
-
|
|
271
|
+
GeminiVideoModelSizeByName[TModel],
|
|
268
272
|
GeminiVideoModelDurationByName[TModel]
|
|
269
273
|
>,
|
|
270
274
|
): Promise<VideoJobResult> {
|
|
@@ -339,7 +343,7 @@ export class GeminiVideoAdapter<
|
|
|
339
343
|
private async createInteractionsVideoJob(
|
|
340
344
|
options: VideoGenerationOptions<
|
|
341
345
|
GeminiVideoModelProviderOptionsByName[TModel],
|
|
342
|
-
|
|
346
|
+
GeminiVideoModelSizeByName[TModel],
|
|
343
347
|
GeminiVideoModelDurationByName[TModel]
|
|
344
348
|
>,
|
|
345
349
|
): Promise<VideoJobResult> {
|
|
@@ -384,16 +388,23 @@ export class GeminiVideoAdapter<
|
|
|
384
388
|
)
|
|
385
389
|
}
|
|
386
390
|
|
|
387
|
-
// Aspect ratio
|
|
388
|
-
// a `"<seconds>s"` string
|
|
389
|
-
// (
|
|
390
|
-
//
|
|
391
|
+
// Aspect ratio, clip length, and resolution ride on `response_format`.
|
|
392
|
+
// Duration is a `"<seconds>s"` string. Resolution comes from the
|
|
393
|
+
// optional `size` suffix (`'16:9_1080p'`), defaulting to 720p when
|
|
394
|
+
// omitted — https://ai.google.dev/gemini-api/docs/omni
|
|
395
|
+
const parsedSize =
|
|
396
|
+
size !== undefined ? parseGeminiOmniVideoSize(size) : undefined
|
|
391
397
|
const responseFormat =
|
|
392
|
-
|
|
398
|
+
parsedSize !== undefined || duration !== undefined
|
|
393
399
|
? {
|
|
394
400
|
response_format: {
|
|
395
401
|
type: 'video' as const,
|
|
396
|
-
...(
|
|
402
|
+
...(parsedSize !== undefined && {
|
|
403
|
+
aspect_ratio: parsedSize.aspectRatio,
|
|
404
|
+
...(parsedSize.resolution !== undefined && {
|
|
405
|
+
resolution: parsedSize.resolution,
|
|
406
|
+
}),
|
|
407
|
+
}),
|
|
397
408
|
...(duration !== undefined && { duration: `${duration}s` }),
|
|
398
409
|
},
|
|
399
410
|
}
|
|
@@ -659,6 +670,12 @@ export class GeminiVideoAdapter<
|
|
|
659
670
|
}
|
|
660
671
|
}
|
|
661
672
|
|
|
673
|
+
/** @deprecated Shuts down 2026-09-30. Use `gemini-omni-1.1-flash`. */
|
|
674
|
+
export function createGeminiVideo(
|
|
675
|
+
model: 'gemini-omni-flash-preview',
|
|
676
|
+
apiKey: string,
|
|
677
|
+
config?: Omit<GeminiVideoConfig, 'apiKey'>,
|
|
678
|
+
): GeminiVideoAdapter<'gemini-omni-flash-preview'>
|
|
662
679
|
/**
|
|
663
680
|
* Creates a Gemini video adapter with an explicit API key.
|
|
664
681
|
* Type resolution happens here at the call site.
|
|
@@ -681,6 +698,11 @@ export class GeminiVideoAdapter<
|
|
|
681
698
|
* });
|
|
682
699
|
* ```
|
|
683
700
|
*/
|
|
701
|
+
export function createGeminiVideo<TModel extends GeminiVideoModel>(
|
|
702
|
+
model: TModel,
|
|
703
|
+
apiKey: string,
|
|
704
|
+
config?: Omit<GeminiVideoConfig, 'apiKey'>,
|
|
705
|
+
): GeminiVideoAdapter<TModel>
|
|
684
706
|
export function createGeminiVideo<TModel extends GeminiVideoModel>(
|
|
685
707
|
model: TModel,
|
|
686
708
|
apiKey: string,
|
|
@@ -689,6 +711,11 @@ export function createGeminiVideo<TModel extends GeminiVideoModel>(
|
|
|
689
711
|
return new GeminiVideoAdapter({ apiKey, ...config }, model)
|
|
690
712
|
}
|
|
691
713
|
|
|
714
|
+
/** @deprecated Shuts down 2026-09-30. Use `gemini-omni-1.1-flash`. */
|
|
715
|
+
export function geminiVideo(
|
|
716
|
+
model: 'gemini-omni-flash-preview',
|
|
717
|
+
config?: Omit<GeminiVideoConfig, 'apiKey'>,
|
|
718
|
+
): GeminiVideoAdapter<'gemini-omni-flash-preview'>
|
|
692
719
|
/**
|
|
693
720
|
* Creates a Gemini video adapter with automatic API key detection from environment variables.
|
|
694
721
|
* Type resolution happens here at the call site.
|
|
@@ -719,6 +746,10 @@ export function createGeminiVideo<TModel extends GeminiVideoModel>(
|
|
|
719
746
|
* const status = await getVideoJobStatus({ adapter, jobId });
|
|
720
747
|
* ```
|
|
721
748
|
*/
|
|
749
|
+
export function geminiVideo<TModel extends GeminiVideoModel>(
|
|
750
|
+
model: TModel,
|
|
751
|
+
config?: Omit<GeminiVideoConfig, 'apiKey'>,
|
|
752
|
+
): GeminiVideoAdapter<TModel>
|
|
722
753
|
export function geminiVideo<TModel extends GeminiVideoModel>(
|
|
723
754
|
model: TModel,
|
|
724
755
|
config?: Omit<GeminiVideoConfig, 'apiKey'>,
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
import { FileState } from '@google/genai'
|
|
2
|
+
import { createGeminiClient, getGeminiApiKeyFromEnv } from '../utils'
|
|
3
|
+
import type { GeminiVideoMetadata } from '../message-types'
|
|
4
|
+
import type { VideoPart } from '@tanstack/ai'
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* A file uploaded to the Gemini Files API and ready to reference in a message.
|
|
8
|
+
*/
|
|
9
|
+
export interface GeminiUploadedFile {
|
|
10
|
+
/** Resource name, e.g. `"files/abc123"`. */
|
|
11
|
+
name: string
|
|
12
|
+
/** File URI to reference from message content (as a `url` source). */
|
|
13
|
+
uri: string
|
|
14
|
+
/** MIME type reported by the Files API. */
|
|
15
|
+
mimeType: string
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* Options for {@link uploadGeminiFile}.
|
|
20
|
+
*/
|
|
21
|
+
export interface GeminiUploadFileOptions {
|
|
22
|
+
/**
|
|
23
|
+
* API key. Falls back to `GOOGLE_API_KEY` / `GEMINI_API_KEY` from the
|
|
24
|
+
* environment when omitted.
|
|
25
|
+
*/
|
|
26
|
+
apiKey?: string
|
|
27
|
+
/**
|
|
28
|
+
* MIME type of the file (e.g. `"video/mp4"`). Recommended so the Files API
|
|
29
|
+
* processes and serves the file with the correct type.
|
|
30
|
+
*/
|
|
31
|
+
mimeType?: string
|
|
32
|
+
/** Poll interval while the file is `PROCESSING`, in ms. Default `5000`. */
|
|
33
|
+
pollIntervalMs?: number
|
|
34
|
+
/** Max time to wait for processing, in ms. Default `300000` (5 min). */
|
|
35
|
+
timeoutMs?: number
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Upload a file via the Gemini Files API and wait until it is `ACTIVE`.
|
|
40
|
+
*
|
|
41
|
+
* Large media (notably video) must be uploaded rather than inlined as base64;
|
|
42
|
+
* the Files API processes uploads asynchronously. This wraps the upload +
|
|
43
|
+
* poll-until-ready loop and returns a reference you can drop into message
|
|
44
|
+
* content as a `url` source (see {@link geminiVideoPart}).
|
|
45
|
+
*
|
|
46
|
+
* @throws if the upload has no URI, or processing fails or times out.
|
|
47
|
+
*/
|
|
48
|
+
export async function uploadGeminiFile(
|
|
49
|
+
file: string | Blob,
|
|
50
|
+
options: GeminiUploadFileOptions = {},
|
|
51
|
+
): Promise<GeminiUploadedFile> {
|
|
52
|
+
const {
|
|
53
|
+
apiKey = getGeminiApiKeyFromEnv(),
|
|
54
|
+
mimeType,
|
|
55
|
+
pollIntervalMs = 5000,
|
|
56
|
+
timeoutMs = 300_000,
|
|
57
|
+
} = options
|
|
58
|
+
|
|
59
|
+
const client = createGeminiClient({ apiKey })
|
|
60
|
+
|
|
61
|
+
let uploaded = await client.files.upload({
|
|
62
|
+
file,
|
|
63
|
+
...(mimeType && { config: { mimeType } }),
|
|
64
|
+
})
|
|
65
|
+
|
|
66
|
+
const fileName = uploaded.name
|
|
67
|
+
if (!fileName) {
|
|
68
|
+
throw new Error('Gemini file upload did not return a file name.')
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
const deadline = Date.now() + timeoutMs
|
|
72
|
+
while (uploaded.state === FileState.PROCESSING) {
|
|
73
|
+
if (Date.now() > deadline) {
|
|
74
|
+
throw new Error(
|
|
75
|
+
`Gemini file processing timed out after ${timeoutMs}ms (${fileName}).`,
|
|
76
|
+
)
|
|
77
|
+
}
|
|
78
|
+
await new Promise((resolve) => setTimeout(resolve, pollIntervalMs))
|
|
79
|
+
uploaded = await client.files.get({ name: fileName })
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
if (uploaded.state === FileState.FAILED) {
|
|
83
|
+
throw new Error(
|
|
84
|
+
`Gemini file processing failed: ${uploaded.error?.message ?? String(uploaded.state)}`,
|
|
85
|
+
)
|
|
86
|
+
}
|
|
87
|
+
if (!uploaded.uri) {
|
|
88
|
+
throw new Error('Gemini file upload did not return a URI.')
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
return {
|
|
92
|
+
name: fileName,
|
|
93
|
+
uri: uploaded.uri,
|
|
94
|
+
mimeType: uploaded.mimeType ?? mimeType ?? 'application/octet-stream',
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Build a TanStack AI video content part from an uploaded Gemini file.
|
|
100
|
+
*
|
|
101
|
+
* Pass `metadata` to control understanding — e.g.
|
|
102
|
+
* `{ processing: 'agentic' }` to route through the agentic Interactions path,
|
|
103
|
+
* or `{ fps, startOffset, endOffset }` for single-pass sampling controls.
|
|
104
|
+
*/
|
|
105
|
+
export function geminiVideoPart(
|
|
106
|
+
file: GeminiUploadedFile,
|
|
107
|
+
metadata?: GeminiVideoMetadata,
|
|
108
|
+
): VideoPart<GeminiVideoMetadata> {
|
|
109
|
+
return {
|
|
110
|
+
type: 'video',
|
|
111
|
+
source: { type: 'url', value: file.uri, mimeType: file.mimeType },
|
|
112
|
+
...(metadata && { metadata }),
|
|
113
|
+
}
|
|
114
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -60,6 +60,14 @@ export type {
|
|
|
60
60
|
// having to add `@google/genai` to their own dependencies.
|
|
61
61
|
export { HarmBlockThreshold, HarmCategory } from '@google/genai'
|
|
62
62
|
|
|
63
|
+
// Files API helpers — upload + poll a file (e.g. video) until it is ACTIVE
|
|
64
|
+
export {
|
|
65
|
+
uploadGeminiFile,
|
|
66
|
+
geminiVideoPart,
|
|
67
|
+
type GeminiUploadedFile,
|
|
68
|
+
type GeminiUploadFileOptions,
|
|
69
|
+
} from './files/index'
|
|
70
|
+
|
|
63
71
|
// Embedding adapter - for embedding vectors
|
|
64
72
|
export {
|
|
65
73
|
GeminiEmbeddingAdapter,
|
|
@@ -108,10 +116,13 @@ export {
|
|
|
108
116
|
GEMINI_VIDEO_DURATIONS,
|
|
109
117
|
getGeminiVideoDurationOptions,
|
|
110
118
|
isInteractionsVideoModel,
|
|
119
|
+
parseGeminiOmniVideoSize,
|
|
111
120
|
} from './video/video-provider-options'
|
|
112
121
|
export type {
|
|
113
122
|
GeminiInteractionsVideoModel,
|
|
114
123
|
GeminiOmniVideoProviderOptions,
|
|
124
|
+
GeminiOmniVideoResolution,
|
|
125
|
+
GeminiOmniVideoSize,
|
|
115
126
|
GeminiVideoModel,
|
|
116
127
|
GeminiVideoModelDurationByName,
|
|
117
128
|
GeminiVideoModelInputModalitiesByName,
|
|
@@ -165,6 +176,7 @@ export type {
|
|
|
165
176
|
GeminiImageMetadata,
|
|
166
177
|
GeminiAudioMetadata,
|
|
167
178
|
GeminiVideoMetadata,
|
|
179
|
+
GeminiVideoProcessing,
|
|
168
180
|
GeminiDocumentMetadata,
|
|
169
181
|
GeminiMessageMetadataByModality,
|
|
170
182
|
} from './message-types'
|
package/src/message-types.ts
CHANGED
|
@@ -87,6 +87,19 @@ export interface GeminiAudioMetadata {
|
|
|
87
87
|
mimeType?: GeminiAudioMimeType
|
|
88
88
|
}
|
|
89
89
|
|
|
90
|
+
/**
|
|
91
|
+
* How Gemini processes a video for understanding.
|
|
92
|
+
*
|
|
93
|
+
* - `static` (default): single-pass frame sampling via `generateContent`.
|
|
94
|
+
* - `agentic`: multi-pass "agentic" video understanding, GA on the
|
|
95
|
+
* `agentic_video`-capable flash models (`gemini-3.7-flash`,
|
|
96
|
+
* `gemini-3.6-flash`, `gemini-3.5-flash-lite`). The text adapter routes the
|
|
97
|
+
* request through the Interactions API instead of `generateContent`. With
|
|
98
|
+
* `agentic`, the sampling rate is expressed in the text prompt (e.g. "watch
|
|
99
|
+
* it at 0.5 fps"), not via `fps`.
|
|
100
|
+
*/
|
|
101
|
+
export type GeminiVideoProcessing = 'agentic' | 'static'
|
|
102
|
+
|
|
90
103
|
/**
|
|
91
104
|
* Metadata for Gemini video content parts.
|
|
92
105
|
*/
|
|
@@ -98,6 +111,30 @@ export interface GeminiVideoMetadata {
|
|
|
98
111
|
* @see https://ai.google.dev/gemini-api/docs/vision#video-requirements
|
|
99
112
|
*/
|
|
100
113
|
mimeType?: GeminiVideoMimeType
|
|
114
|
+
/**
|
|
115
|
+
* How the model processes this video for understanding. When set, the
|
|
116
|
+
* adapter routes the request through the Gemini Interactions API. Omit for
|
|
117
|
+
* the default single-pass `generateContent` behavior.
|
|
118
|
+
*/
|
|
119
|
+
processing?: GeminiVideoProcessing
|
|
120
|
+
/**
|
|
121
|
+
* Frame-rate sampling density (frames per second) for single-pass
|
|
122
|
+
* (`generateContent`) understanding. Valid range (0, 24]; defaults to 1.0
|
|
123
|
+
* on the server. Ignored when `processing` is `agentic`.
|
|
124
|
+
*
|
|
125
|
+
* @see https://ai.google.dev/gemini-api/docs/vision#customize-frame-rate
|
|
126
|
+
*/
|
|
127
|
+
fps?: number
|
|
128
|
+
/**
|
|
129
|
+
* Clip start offset, as a decimal number of seconds with an `s` suffix
|
|
130
|
+
* (e.g. `"10.5s"`). Restricts understanding to a segment of the video.
|
|
131
|
+
*/
|
|
132
|
+
startOffset?: string
|
|
133
|
+
/**
|
|
134
|
+
* Clip end offset, as a decimal number of seconds with an `s` suffix
|
|
135
|
+
* (e.g. `"45s"`). Restricts understanding to a segment of the video.
|
|
136
|
+
*/
|
|
137
|
+
endOffset?: string
|
|
101
138
|
}
|
|
102
139
|
|
|
103
140
|
/**
|