@tanstack/ai-gemini 0.26.5 → 0.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -9
- package/dist/esm/adapters/text.d.ts +9 -0
- package/dist/esm/adapters/text.js +165 -13
- package/dist/esm/adapters/text.js.map +1 -1
- package/dist/esm/files/index.d.ts +51 -0
- package/dist/esm/files/index.js +59 -0
- package/dist/esm/files/index.js.map +1 -0
- package/dist/esm/index.d.ts +2 -1
- package/dist/esm/index.js +2 -1
- package/dist/esm/message-types.d.ts +36 -0
- package/dist/esm/model-meta.d.ts +3 -3
- package/dist/esm/model-meta.js +3 -0
- package/dist/esm/model-meta.js.map +1 -1
- package/package.json +1 -1
- package/src/adapters/text.ts +253 -20
- package/src/files/index.ts +114 -0
- package/src/index.ts +9 -0
- package/src/message-types.ts +37 -0
- package/src/model-meta.ts +4 -0
package/src/adapters/text.ts
CHANGED
|
@@ -28,6 +28,7 @@ import type {
|
|
|
28
28
|
GoogleGenAI,
|
|
29
29
|
Part,
|
|
30
30
|
ThinkingLevel,
|
|
31
|
+
VideoMetadata,
|
|
31
32
|
} from '@google/genai'
|
|
32
33
|
import type {
|
|
33
34
|
ContentPart,
|
|
@@ -40,9 +41,113 @@ import type { ExternalTextProviderOptions } from '../text/text-provider-options'
|
|
|
40
41
|
import type {
|
|
41
42
|
GeminiMessageMetadataByModality,
|
|
42
43
|
GeminiToolCallMetadata,
|
|
44
|
+
GeminiVideoMetadata,
|
|
45
|
+
GeminiVideoProcessing,
|
|
43
46
|
} from '../message-types'
|
|
44
47
|
import type { GeminiClientConfig } from '../utils/client'
|
|
45
48
|
|
|
49
|
+
/**
|
|
50
|
+
* Fallback MIME types for URL-sourced media parts that don't specify one.
|
|
51
|
+
*/
|
|
52
|
+
const DEFAULT_MEDIA_MIME_TYPES = {
|
|
53
|
+
image: 'image/jpeg',
|
|
54
|
+
audio: 'audio/mp3',
|
|
55
|
+
video: 'video/mp4',
|
|
56
|
+
document: 'application/pdf',
|
|
57
|
+
} as const
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* Content block shape for an Interactions API `input` step. The installed
|
|
61
|
+
* @google/genai types predate the video `processing` field, so we model the
|
|
62
|
+
* subset we emit and cast at the call site.
|
|
63
|
+
*/
|
|
64
|
+
type InteractionContent =
|
|
65
|
+
| { type: 'text'; text: string }
|
|
66
|
+
| {
|
|
67
|
+
type: 'video'
|
|
68
|
+
uri?: string
|
|
69
|
+
data?: string
|
|
70
|
+
mime_type?: string
|
|
71
|
+
processing?: GeminiVideoProcessing
|
|
72
|
+
}
|
|
73
|
+
| {
|
|
74
|
+
type: 'image' | 'audio' | 'document'
|
|
75
|
+
uri?: string
|
|
76
|
+
data?: string
|
|
77
|
+
mime_type?: string
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
interface InteractionStep {
|
|
81
|
+
type: 'user_input' | 'model_output'
|
|
82
|
+
content: Array<InteractionContent>
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/** True when any message carries a video part requesting agentic processing. */
|
|
86
|
+
function hasAgenticVideo(messages: Array<ModelMessage>): boolean {
|
|
87
|
+
return messages.some(
|
|
88
|
+
(msg) =>
|
|
89
|
+
Array.isArray(msg.content) &&
|
|
90
|
+
msg.content.some(
|
|
91
|
+
(part) =>
|
|
92
|
+
part.type === 'video' &&
|
|
93
|
+
(part.metadata as GeminiVideoMetadata | undefined)?.processing ===
|
|
94
|
+
'agentic',
|
|
95
|
+
),
|
|
96
|
+
)
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/** Convert a single content part to an Interactions API content block. */
|
|
100
|
+
function contentPartToInteraction(part: ContentPart): InteractionContent {
|
|
101
|
+
if (part.type === 'text') {
|
|
102
|
+
return { type: 'text', text: part.content }
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
const source = part.source
|
|
106
|
+
const mimeType =
|
|
107
|
+
source.type === 'data'
|
|
108
|
+
? source.mimeType
|
|
109
|
+
: (source.mimeType ?? DEFAULT_MEDIA_MIME_TYPES[part.type])
|
|
110
|
+
const base =
|
|
111
|
+
source.type === 'data'
|
|
112
|
+
? { data: source.value, mime_type: mimeType }
|
|
113
|
+
: { uri: source.value, mime_type: mimeType }
|
|
114
|
+
|
|
115
|
+
if (part.type === 'video') {
|
|
116
|
+
const processing = (part.metadata as GeminiVideoMetadata | undefined)
|
|
117
|
+
?.processing
|
|
118
|
+
return { type: 'video', ...base, ...(processing && { processing }) }
|
|
119
|
+
}
|
|
120
|
+
return { type: part.type, ...base }
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Build the Interactions API `input` from chat messages. Each user/assistant
|
|
125
|
+
* message becomes a `user_input` / `model_output` step wrapping its content
|
|
126
|
+
* blocks — the wrapping the Python SDK performs implicitly but the JS SDK
|
|
127
|
+
* does not. Tool messages are skipped (unsupported on this path).
|
|
128
|
+
*/
|
|
129
|
+
function buildInteractionsInput(
|
|
130
|
+
messages: Array<ModelMessage>,
|
|
131
|
+
): Array<InteractionStep> {
|
|
132
|
+
const steps: Array<InteractionStep> = []
|
|
133
|
+
for (const msg of messages) {
|
|
134
|
+
if (msg.role === 'tool') continue
|
|
135
|
+
const stepType = msg.role === 'assistant' ? 'model_output' : 'user_input'
|
|
136
|
+
const content: Array<InteractionContent> = []
|
|
137
|
+
if (Array.isArray(msg.content)) {
|
|
138
|
+
for (const part of msg.content) {
|
|
139
|
+
content.push(contentPartToInteraction(part))
|
|
140
|
+
}
|
|
141
|
+
} else if (msg.content) {
|
|
142
|
+
content.push({ type: 'text', text: msg.content })
|
|
143
|
+
}
|
|
144
|
+
if (content.length > 0) {
|
|
145
|
+
steps.push({ type: stepType, content })
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
return steps
|
|
149
|
+
}
|
|
150
|
+
|
|
46
151
|
/**
|
|
47
152
|
* Configuration for Gemini text adapter
|
|
48
153
|
*/
|
|
@@ -122,6 +227,13 @@ export class GeminiTextAdapter<
|
|
|
122
227
|
async *chatStream(
|
|
123
228
|
options: TextOptions<GeminiTextProviderOptions>,
|
|
124
229
|
): AsyncIterable<AdapterYieldChunk> {
|
|
230
|
+
// Agentic video understanding is only exposed through the Interactions API,
|
|
231
|
+
// not generateContent. Detect it and take that path instead.
|
|
232
|
+
if (hasAgenticVideo(options.messages)) {
|
|
233
|
+
yield* this.interactionsStream(options)
|
|
234
|
+
return
|
|
235
|
+
}
|
|
236
|
+
|
|
125
237
|
const mappedOptions = this.mapCommonOptionsToGemini(options)
|
|
126
238
|
const { logger } = options
|
|
127
239
|
|
|
@@ -161,6 +273,114 @@ export class GeminiTextAdapter<
|
|
|
161
273
|
}
|
|
162
274
|
}
|
|
163
275
|
|
|
276
|
+
/**
|
|
277
|
+
* Agentic video-understanding path via the Interactions API.
|
|
278
|
+
*
|
|
279
|
+
* The Interactions API (unlike `generateContent`) requires message parts to
|
|
280
|
+
* be wrapped in `user_input` / `model_output` steps, and it accepts the
|
|
281
|
+
* `processing: 'agentic'` video flag. This is a non-streaming call whose
|
|
282
|
+
* single text result is re-emitted as AG-UI stream chunks.
|
|
283
|
+
*/
|
|
284
|
+
private async *interactionsStream(
|
|
285
|
+
options: TextOptions<GeminiTextProviderOptions>,
|
|
286
|
+
): AsyncIterable<AdapterYieldChunk> {
|
|
287
|
+
const model = options.model
|
|
288
|
+
const { logger } = options
|
|
289
|
+
const runId = options.runId ?? generateId(this.name)
|
|
290
|
+
const threadId = options.threadId ?? generateId(this.name)
|
|
291
|
+
const messageId = generateId(this.name)
|
|
292
|
+
|
|
293
|
+
try {
|
|
294
|
+
logger.request(
|
|
295
|
+
`activity=chat provider=gemini model=${model} messages=${options.messages.length} mode=interactions-agentic-video`,
|
|
296
|
+
{ provider: 'gemini', model },
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
const normalizedPrompts = normalizeSystemPrompts(options.systemPrompts)
|
|
300
|
+
const systemInstruction =
|
|
301
|
+
normalizedPrompts.length > 0
|
|
302
|
+
? normalizedPrompts.map((p) => p.content).join('\n')
|
|
303
|
+
: undefined
|
|
304
|
+
|
|
305
|
+
const input = buildInteractionsInput(options.messages)
|
|
306
|
+
|
|
307
|
+
// The installed @google/genai (2.10.0) Interactions `VideoContent` type
|
|
308
|
+
// predates the `processing` field, so the structurally-built input is
|
|
309
|
+
// cast at the call boundary. The SDK forwards it to the wire unchanged.
|
|
310
|
+
const interaction = await this.client.interactions.create({
|
|
311
|
+
model,
|
|
312
|
+
...(systemInstruction !== undefined && {
|
|
313
|
+
system_instruction: systemInstruction,
|
|
314
|
+
}),
|
|
315
|
+
input: input as never,
|
|
316
|
+
})
|
|
317
|
+
|
|
318
|
+
const text = interaction.output_text ?? ''
|
|
319
|
+
|
|
320
|
+
yield {
|
|
321
|
+
type: EventType.RUN_STARTED,
|
|
322
|
+
runId,
|
|
323
|
+
threadId,
|
|
324
|
+
model,
|
|
325
|
+
timestamp: Date.now(),
|
|
326
|
+
parentRunId: options.parentRunId,
|
|
327
|
+
}
|
|
328
|
+
yield {
|
|
329
|
+
type: EventType.TEXT_MESSAGE_START,
|
|
330
|
+
messageId,
|
|
331
|
+
model,
|
|
332
|
+
timestamp: Date.now(),
|
|
333
|
+
role: 'assistant',
|
|
334
|
+
}
|
|
335
|
+
if (text) {
|
|
336
|
+
yield {
|
|
337
|
+
type: EventType.TEXT_MESSAGE_CONTENT,
|
|
338
|
+
messageId,
|
|
339
|
+
model,
|
|
340
|
+
timestamp: Date.now(),
|
|
341
|
+
delta: text,
|
|
342
|
+
content: text,
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
yield {
|
|
346
|
+
type: EventType.TEXT_MESSAGE_END,
|
|
347
|
+
messageId,
|
|
348
|
+
model,
|
|
349
|
+
timestamp: Date.now(),
|
|
350
|
+
}
|
|
351
|
+
yield {
|
|
352
|
+
type: EventType.RUN_FINISHED,
|
|
353
|
+
runId,
|
|
354
|
+
threadId,
|
|
355
|
+
model,
|
|
356
|
+
timestamp: Date.now(),
|
|
357
|
+
finishReason: 'stop',
|
|
358
|
+
}
|
|
359
|
+
} catch (error) {
|
|
360
|
+
const rawEvent = toRunErrorRawEvent(error)
|
|
361
|
+
logger.errors('gemini.interactionsStream fatal', {
|
|
362
|
+
error,
|
|
363
|
+
source: 'gemini.interactionsStream',
|
|
364
|
+
})
|
|
365
|
+
yield {
|
|
366
|
+
type: EventType.RUN_ERROR,
|
|
367
|
+
model,
|
|
368
|
+
timestamp: Date.now(),
|
|
369
|
+
message:
|
|
370
|
+
error instanceof Error
|
|
371
|
+
? error.message
|
|
372
|
+
: 'An unknown error occurred during the chat stream.',
|
|
373
|
+
...(rawEvent !== undefined && { rawEvent }),
|
|
374
|
+
error: {
|
|
375
|
+
message:
|
|
376
|
+
error instanceof Error
|
|
377
|
+
? error.message
|
|
378
|
+
: 'An unknown error occurred during the chat stream.',
|
|
379
|
+
},
|
|
380
|
+
}
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
|
|
164
384
|
/**
|
|
165
385
|
* Generate structured output using Gemini's native JSON response format.
|
|
166
386
|
* Uses responseMimeType: 'application/json' and responseSchema for structured output.
|
|
@@ -595,29 +815,42 @@ export class GeminiTextAdapter<
|
|
|
595
815
|
case 'audio':
|
|
596
816
|
case 'video':
|
|
597
817
|
case 'document': {
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
818
|
+
const geminiPart: Part =
|
|
819
|
+
part.source.type === 'data'
|
|
820
|
+
? {
|
|
821
|
+
inlineData: {
|
|
822
|
+
data: part.source.value,
|
|
823
|
+
mimeType: part.source.mimeType,
|
|
824
|
+
},
|
|
825
|
+
}
|
|
826
|
+
: {
|
|
827
|
+
fileData: {
|
|
828
|
+
fileUri: part.source.value,
|
|
829
|
+
// For URL sources, use provided mimeType or fall back to
|
|
830
|
+
// reasonable defaults.
|
|
831
|
+
mimeType:
|
|
832
|
+
part.source.mimeType ?? DEFAULT_MEDIA_MIME_TYPES[part.type],
|
|
833
|
+
},
|
|
834
|
+
}
|
|
835
|
+
|
|
836
|
+
// Apply single-pass video sampling controls (fps / clip offsets) from
|
|
837
|
+
// the part metadata. `processing: 'agentic'` is handled separately via
|
|
838
|
+
// the Interactions API and never reaches this generateContent path.
|
|
839
|
+
if (part.type === 'video') {
|
|
840
|
+
const meta = part.metadata as GeminiVideoMetadata | undefined
|
|
841
|
+
const videoMetadata: VideoMetadata = {
|
|
842
|
+
...(meta?.fps !== undefined && { fps: meta.fps }),
|
|
843
|
+
...(meta?.startOffset !== undefined && {
|
|
844
|
+
startOffset: meta.startOffset,
|
|
845
|
+
}),
|
|
846
|
+
...(meta?.endOffset !== undefined && { endOffset: meta.endOffset }),
|
|
604
847
|
}
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
const defaultMimeType = {
|
|
608
|
-
image: 'image/jpeg',
|
|
609
|
-
audio: 'audio/mp3',
|
|
610
|
-
video: 'video/mp4',
|
|
611
|
-
document: 'application/pdf',
|
|
612
|
-
}[part.type]
|
|
613
|
-
|
|
614
|
-
return {
|
|
615
|
-
fileData: {
|
|
616
|
-
fileUri: part.source.value,
|
|
617
|
-
mimeType: part.source.mimeType ?? defaultMimeType,
|
|
618
|
-
},
|
|
848
|
+
if (Object.keys(videoMetadata).length > 0) {
|
|
849
|
+
geminiPart.videoMetadata = videoMetadata
|
|
619
850
|
}
|
|
620
851
|
}
|
|
852
|
+
|
|
853
|
+
return geminiPart
|
|
621
854
|
}
|
|
622
855
|
default: {
|
|
623
856
|
const _exhaustiveCheck: never = part
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
import { FileState } from '@google/genai'
|
|
2
|
+
import { createGeminiClient, getGeminiApiKeyFromEnv } from '../utils'
|
|
3
|
+
import type { GeminiVideoMetadata } from '../message-types'
|
|
4
|
+
import type { VideoPart } from '@tanstack/ai'
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* A file uploaded to the Gemini Files API and ready to reference in a message.
|
|
8
|
+
*/
|
|
9
|
+
export interface GeminiUploadedFile {
|
|
10
|
+
/** Resource name, e.g. `"files/abc123"`. */
|
|
11
|
+
name: string
|
|
12
|
+
/** File URI to reference from message content (as a `url` source). */
|
|
13
|
+
uri: string
|
|
14
|
+
/** MIME type reported by the Files API. */
|
|
15
|
+
mimeType: string
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* Options for {@link uploadGeminiFile}.
|
|
20
|
+
*/
|
|
21
|
+
export interface GeminiUploadFileOptions {
|
|
22
|
+
/**
|
|
23
|
+
* API key. Falls back to `GOOGLE_API_KEY` / `GEMINI_API_KEY` from the
|
|
24
|
+
* environment when omitted.
|
|
25
|
+
*/
|
|
26
|
+
apiKey?: string
|
|
27
|
+
/**
|
|
28
|
+
* MIME type of the file (e.g. `"video/mp4"`). Recommended so the Files API
|
|
29
|
+
* processes and serves the file with the correct type.
|
|
30
|
+
*/
|
|
31
|
+
mimeType?: string
|
|
32
|
+
/** Poll interval while the file is `PROCESSING`, in ms. Default `5000`. */
|
|
33
|
+
pollIntervalMs?: number
|
|
34
|
+
/** Max time to wait for processing, in ms. Default `300000` (5 min). */
|
|
35
|
+
timeoutMs?: number
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Upload a file via the Gemini Files API and wait until it is `ACTIVE`.
|
|
40
|
+
*
|
|
41
|
+
* Large media (notably video) must be uploaded rather than inlined as base64;
|
|
42
|
+
* the Files API processes uploads asynchronously. This wraps the upload +
|
|
43
|
+
* poll-until-ready loop and returns a reference you can drop into message
|
|
44
|
+
* content as a `url` source (see {@link geminiVideoPart}).
|
|
45
|
+
*
|
|
46
|
+
* @throws if the upload has no URI, or processing fails or times out.
|
|
47
|
+
*/
|
|
48
|
+
export async function uploadGeminiFile(
|
|
49
|
+
file: string | Blob,
|
|
50
|
+
options: GeminiUploadFileOptions = {},
|
|
51
|
+
): Promise<GeminiUploadedFile> {
|
|
52
|
+
const {
|
|
53
|
+
apiKey = getGeminiApiKeyFromEnv(),
|
|
54
|
+
mimeType,
|
|
55
|
+
pollIntervalMs = 5000,
|
|
56
|
+
timeoutMs = 300_000,
|
|
57
|
+
} = options
|
|
58
|
+
|
|
59
|
+
const client = createGeminiClient({ apiKey })
|
|
60
|
+
|
|
61
|
+
let uploaded = await client.files.upload({
|
|
62
|
+
file,
|
|
63
|
+
...(mimeType && { config: { mimeType } }),
|
|
64
|
+
})
|
|
65
|
+
|
|
66
|
+
const fileName = uploaded.name
|
|
67
|
+
if (!fileName) {
|
|
68
|
+
throw new Error('Gemini file upload did not return a file name.')
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
const deadline = Date.now() + timeoutMs
|
|
72
|
+
while (uploaded.state === FileState.PROCESSING) {
|
|
73
|
+
if (Date.now() > deadline) {
|
|
74
|
+
throw new Error(
|
|
75
|
+
`Gemini file processing timed out after ${timeoutMs}ms (${fileName}).`,
|
|
76
|
+
)
|
|
77
|
+
}
|
|
78
|
+
await new Promise((resolve) => setTimeout(resolve, pollIntervalMs))
|
|
79
|
+
uploaded = await client.files.get({ name: fileName })
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
if (uploaded.state === FileState.FAILED) {
|
|
83
|
+
throw new Error(
|
|
84
|
+
`Gemini file processing failed: ${uploaded.error?.message ?? String(uploaded.state)}`,
|
|
85
|
+
)
|
|
86
|
+
}
|
|
87
|
+
if (!uploaded.uri) {
|
|
88
|
+
throw new Error('Gemini file upload did not return a URI.')
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
return {
|
|
92
|
+
name: fileName,
|
|
93
|
+
uri: uploaded.uri,
|
|
94
|
+
mimeType: uploaded.mimeType ?? mimeType ?? 'application/octet-stream',
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Build a TanStack AI video content part from an uploaded Gemini file.
|
|
100
|
+
*
|
|
101
|
+
* Pass `metadata` to control understanding — e.g.
|
|
102
|
+
* `{ processing: 'agentic' }` to route through the agentic Interactions path,
|
|
103
|
+
* or `{ fps, startOffset, endOffset }` for single-pass sampling controls.
|
|
104
|
+
*/
|
|
105
|
+
export function geminiVideoPart(
|
|
106
|
+
file: GeminiUploadedFile,
|
|
107
|
+
metadata?: GeminiVideoMetadata,
|
|
108
|
+
): VideoPart<GeminiVideoMetadata> {
|
|
109
|
+
return {
|
|
110
|
+
type: 'video',
|
|
111
|
+
source: { type: 'url', value: file.uri, mimeType: file.mimeType },
|
|
112
|
+
...(metadata && { metadata }),
|
|
113
|
+
}
|
|
114
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -60,6 +60,14 @@ export type {
|
|
|
60
60
|
// having to add `@google/genai` to their own dependencies.
|
|
61
61
|
export { HarmBlockThreshold, HarmCategory } from '@google/genai'
|
|
62
62
|
|
|
63
|
+
// Files API helpers — upload + poll a file (e.g. video) until it is ACTIVE
|
|
64
|
+
export {
|
|
65
|
+
uploadGeminiFile,
|
|
66
|
+
geminiVideoPart,
|
|
67
|
+
type GeminiUploadedFile,
|
|
68
|
+
type GeminiUploadFileOptions,
|
|
69
|
+
} from './files/index'
|
|
70
|
+
|
|
63
71
|
// Embedding adapter - for embedding vectors
|
|
64
72
|
export {
|
|
65
73
|
GeminiEmbeddingAdapter,
|
|
@@ -168,6 +176,7 @@ export type {
|
|
|
168
176
|
GeminiImageMetadata,
|
|
169
177
|
GeminiAudioMetadata,
|
|
170
178
|
GeminiVideoMetadata,
|
|
179
|
+
GeminiVideoProcessing,
|
|
171
180
|
GeminiDocumentMetadata,
|
|
172
181
|
GeminiMessageMetadataByModality,
|
|
173
182
|
} from './message-types'
|
package/src/message-types.ts
CHANGED
|
@@ -87,6 +87,19 @@ export interface GeminiAudioMetadata {
|
|
|
87
87
|
mimeType?: GeminiAudioMimeType
|
|
88
88
|
}
|
|
89
89
|
|
|
90
|
+
/**
|
|
91
|
+
* How Gemini processes a video for understanding.
|
|
92
|
+
*
|
|
93
|
+
* - `static` (default): single-pass frame sampling via `generateContent`.
|
|
94
|
+
* - `agentic`: multi-pass "agentic" video understanding, GA on the
|
|
95
|
+
* `agentic_video`-capable flash models (`gemini-3.7-flash`,
|
|
96
|
+
* `gemini-3.6-flash`, `gemini-3.5-flash-lite`). The text adapter routes the
|
|
97
|
+
* request through the Interactions API instead of `generateContent`. With
|
|
98
|
+
* `agentic`, the sampling rate is expressed in the text prompt (e.g. "watch
|
|
99
|
+
* it at 0.5 fps"), not via `fps`.
|
|
100
|
+
*/
|
|
101
|
+
export type GeminiVideoProcessing = 'agentic' | 'static'
|
|
102
|
+
|
|
90
103
|
/**
|
|
91
104
|
* Metadata for Gemini video content parts.
|
|
92
105
|
*/
|
|
@@ -98,6 +111,30 @@ export interface GeminiVideoMetadata {
|
|
|
98
111
|
* @see https://ai.google.dev/gemini-api/docs/vision#video-requirements
|
|
99
112
|
*/
|
|
100
113
|
mimeType?: GeminiVideoMimeType
|
|
114
|
+
/**
|
|
115
|
+
* How the model processes this video for understanding. When set, the
|
|
116
|
+
* adapter routes the request through the Gemini Interactions API. Omit for
|
|
117
|
+
* the default single-pass `generateContent` behavior.
|
|
118
|
+
*/
|
|
119
|
+
processing?: GeminiVideoProcessing
|
|
120
|
+
/**
|
|
121
|
+
* Frame-rate sampling density (frames per second) for single-pass
|
|
122
|
+
* (`generateContent`) understanding. Valid range (0, 24]; defaults to 1.0
|
|
123
|
+
* on the server. Ignored when `processing` is `agentic`.
|
|
124
|
+
*
|
|
125
|
+
* @see https://ai.google.dev/gemini-api/docs/vision#customize-frame-rate
|
|
126
|
+
*/
|
|
127
|
+
fps?: number
|
|
128
|
+
/**
|
|
129
|
+
* Clip start offset, as a decimal number of seconds with an `s` suffix
|
|
130
|
+
* (e.g. `"10.5s"`). Restricts understanding to a segment of the video.
|
|
131
|
+
*/
|
|
132
|
+
startOffset?: string
|
|
133
|
+
/**
|
|
134
|
+
* Clip end offset, as a decimal number of seconds with an `s` suffix
|
|
135
|
+
* (e.g. `"45s"`). Restricts understanding to a segment of the video.
|
|
136
|
+
*/
|
|
137
|
+
endOffset?: string
|
|
101
138
|
}
|
|
102
139
|
|
|
103
140
|
/**
|
package/src/model-meta.ts
CHANGED
|
@@ -14,6 +14,7 @@ interface ModelMeta<TProviderOptions = unknown> {
|
|
|
14
14
|
input: Array<'text' | 'image' | 'audio' | 'video' | 'document'>
|
|
15
15
|
output: Array<'text' | 'image' | 'audio' | 'video'>
|
|
16
16
|
capabilities?: Array<
|
|
17
|
+
| 'agentic_video'
|
|
17
18
|
| 'audio_generation'
|
|
18
19
|
| 'batch_api'
|
|
19
20
|
| 'caching'
|
|
@@ -879,6 +880,7 @@ const GEMINI_3_7_FLASH = {
|
|
|
879
880
|
input: ['text', 'image', 'video', 'audio', 'document'],
|
|
880
881
|
output: ['text'],
|
|
881
882
|
capabilities: [
|
|
883
|
+
'agentic_video',
|
|
882
884
|
'batch_api',
|
|
883
885
|
'caching',
|
|
884
886
|
'function_calling',
|
|
@@ -923,6 +925,7 @@ const GEMINI_3_6_FLASH = {
|
|
|
923
925
|
input: ['text', 'image', 'video', 'audio', 'document'],
|
|
924
926
|
output: ['text'],
|
|
925
927
|
capabilities: [
|
|
928
|
+
'agentic_video',
|
|
926
929
|
'batch_api',
|
|
927
930
|
'caching',
|
|
928
931
|
'function_calling',
|
|
@@ -1006,6 +1009,7 @@ const GEMINI_3_5_FLASH_LITE = {
|
|
|
1006
1009
|
input: ['text', 'image', 'video', 'audio', 'document'],
|
|
1007
1010
|
output: ['text'],
|
|
1008
1011
|
capabilities: [
|
|
1012
|
+
'agentic_video',
|
|
1009
1013
|
'batch_api',
|
|
1010
1014
|
'caching',
|
|
1011
1015
|
'function_calling',
|