@tanstack/ai-gemini 0.26.4 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -28,6 +28,7 @@ import type {
28
28
  GoogleGenAI,
29
29
  Part,
30
30
  ThinkingLevel,
31
+ VideoMetadata,
31
32
  } from '@google/genai'
32
33
  import type {
33
34
  ContentPart,
@@ -40,9 +41,113 @@ import type { ExternalTextProviderOptions } from '../text/text-provider-options'
40
41
  import type {
41
42
  GeminiMessageMetadataByModality,
42
43
  GeminiToolCallMetadata,
44
+ GeminiVideoMetadata,
45
+ GeminiVideoProcessing,
43
46
  } from '../message-types'
44
47
  import type { GeminiClientConfig } from '../utils/client'
45
48
 
49
+ /**
50
+ * Fallback MIME types for URL-sourced media parts that don't specify one.
51
+ */
52
+ const DEFAULT_MEDIA_MIME_TYPES = {
53
+ image: 'image/jpeg',
54
+ audio: 'audio/mp3',
55
+ video: 'video/mp4',
56
+ document: 'application/pdf',
57
+ } as const
58
+
59
+ /**
60
+ * Content block shape for an Interactions API `input` step. The installed
61
+ * @google/genai types predate the video `processing` field, so we model the
62
+ * subset we emit and cast at the call site.
63
+ */
64
+ type InteractionContent =
65
+ | { type: 'text'; text: string }
66
+ | {
67
+ type: 'video'
68
+ uri?: string
69
+ data?: string
70
+ mime_type?: string
71
+ processing?: GeminiVideoProcessing
72
+ }
73
+ | {
74
+ type: 'image' | 'audio' | 'document'
75
+ uri?: string
76
+ data?: string
77
+ mime_type?: string
78
+ }
79
+
80
+ interface InteractionStep {
81
+ type: 'user_input' | 'model_output'
82
+ content: Array<InteractionContent>
83
+ }
84
+
85
+ /** True when any message carries a video part requesting agentic processing. */
86
+ function hasAgenticVideo(messages: Array<ModelMessage>): boolean {
87
+ return messages.some(
88
+ (msg) =>
89
+ Array.isArray(msg.content) &&
90
+ msg.content.some(
91
+ (part) =>
92
+ part.type === 'video' &&
93
+ (part.metadata as GeminiVideoMetadata | undefined)?.processing ===
94
+ 'agentic',
95
+ ),
96
+ )
97
+ }
98
+
99
+ /** Convert a single content part to an Interactions API content block. */
100
+ function contentPartToInteraction(part: ContentPart): InteractionContent {
101
+ if (part.type === 'text') {
102
+ return { type: 'text', text: part.content }
103
+ }
104
+
105
+ const source = part.source
106
+ const mimeType =
107
+ source.type === 'data'
108
+ ? source.mimeType
109
+ : (source.mimeType ?? DEFAULT_MEDIA_MIME_TYPES[part.type])
110
+ const base =
111
+ source.type === 'data'
112
+ ? { data: source.value, mime_type: mimeType }
113
+ : { uri: source.value, mime_type: mimeType }
114
+
115
+ if (part.type === 'video') {
116
+ const processing = (part.metadata as GeminiVideoMetadata | undefined)
117
+ ?.processing
118
+ return { type: 'video', ...base, ...(processing && { processing }) }
119
+ }
120
+ return { type: part.type, ...base }
121
+ }
122
+
123
+ /**
124
+ * Build the Interactions API `input` from chat messages. Each user/assistant
125
+ * message becomes a `user_input` / `model_output` step wrapping its content
126
+ * blocks — the wrapping the Python SDK performs implicitly but the JS SDK
127
+ * does not. Tool messages are skipped (unsupported on this path).
128
+ */
129
+ function buildInteractionsInput(
130
+ messages: Array<ModelMessage>,
131
+ ): Array<InteractionStep> {
132
+ const steps: Array<InteractionStep> = []
133
+ for (const msg of messages) {
134
+ if (msg.role === 'tool') continue
135
+ const stepType = msg.role === 'assistant' ? 'model_output' : 'user_input'
136
+ const content: Array<InteractionContent> = []
137
+ if (Array.isArray(msg.content)) {
138
+ for (const part of msg.content) {
139
+ content.push(contentPartToInteraction(part))
140
+ }
141
+ } else if (msg.content) {
142
+ content.push({ type: 'text', text: msg.content })
143
+ }
144
+ if (content.length > 0) {
145
+ steps.push({ type: stepType, content })
146
+ }
147
+ }
148
+ return steps
149
+ }
150
+
46
151
  /**
47
152
  * Configuration for Gemini text adapter
48
153
  */
@@ -122,6 +227,13 @@ export class GeminiTextAdapter<
122
227
  async *chatStream(
123
228
  options: TextOptions<GeminiTextProviderOptions>,
124
229
  ): AsyncIterable<AdapterYieldChunk> {
230
+ // Agentic video understanding is only exposed through the Interactions API,
231
+ // not generateContent. Detect it and take that path instead.
232
+ if (hasAgenticVideo(options.messages)) {
233
+ yield* this.interactionsStream(options)
234
+ return
235
+ }
236
+
125
237
  const mappedOptions = this.mapCommonOptionsToGemini(options)
126
238
  const { logger } = options
127
239
 
@@ -161,6 +273,114 @@ export class GeminiTextAdapter<
161
273
  }
162
274
  }
163
275
 
276
+ /**
277
+ * Agentic video-understanding path via the Interactions API.
278
+ *
279
+ * The Interactions API (unlike `generateContent`) requires message parts to
280
+ * be wrapped in `user_input` / `model_output` steps, and it accepts the
281
+ * `processing: 'agentic'` video flag. This is a non-streaming call whose
282
+ * single text result is re-emitted as AG-UI stream chunks.
283
+ */
284
+ private async *interactionsStream(
285
+ options: TextOptions<GeminiTextProviderOptions>,
286
+ ): AsyncIterable<AdapterYieldChunk> {
287
+ const model = options.model
288
+ const { logger } = options
289
+ const runId = options.runId ?? generateId(this.name)
290
+ const threadId = options.threadId ?? generateId(this.name)
291
+ const messageId = generateId(this.name)
292
+
293
+ try {
294
+ logger.request(
295
+ `activity=chat provider=gemini model=${model} messages=${options.messages.length} mode=interactions-agentic-video`,
296
+ { provider: 'gemini', model },
297
+ )
298
+
299
+ const normalizedPrompts = normalizeSystemPrompts(options.systemPrompts)
300
+ const systemInstruction =
301
+ normalizedPrompts.length > 0
302
+ ? normalizedPrompts.map((p) => p.content).join('\n')
303
+ : undefined
304
+
305
+ const input = buildInteractionsInput(options.messages)
306
+
307
+ // The installed @google/genai (2.10.0) Interactions `VideoContent` type
308
+ // predates the `processing` field, so the structurally-built input is
309
+ // cast at the call boundary. The SDK forwards it to the wire unchanged.
310
+ const interaction = await this.client.interactions.create({
311
+ model,
312
+ ...(systemInstruction !== undefined && {
313
+ system_instruction: systemInstruction,
314
+ }),
315
+ input: input as never,
316
+ })
317
+
318
+ const text = interaction.output_text ?? ''
319
+
320
+ yield {
321
+ type: EventType.RUN_STARTED,
322
+ runId,
323
+ threadId,
324
+ model,
325
+ timestamp: Date.now(),
326
+ parentRunId: options.parentRunId,
327
+ }
328
+ yield {
329
+ type: EventType.TEXT_MESSAGE_START,
330
+ messageId,
331
+ model,
332
+ timestamp: Date.now(),
333
+ role: 'assistant',
334
+ }
335
+ if (text) {
336
+ yield {
337
+ type: EventType.TEXT_MESSAGE_CONTENT,
338
+ messageId,
339
+ model,
340
+ timestamp: Date.now(),
341
+ delta: text,
342
+ content: text,
343
+ }
344
+ }
345
+ yield {
346
+ type: EventType.TEXT_MESSAGE_END,
347
+ messageId,
348
+ model,
349
+ timestamp: Date.now(),
350
+ }
351
+ yield {
352
+ type: EventType.RUN_FINISHED,
353
+ runId,
354
+ threadId,
355
+ model,
356
+ timestamp: Date.now(),
357
+ finishReason: 'stop',
358
+ }
359
+ } catch (error) {
360
+ const rawEvent = toRunErrorRawEvent(error)
361
+ logger.errors('gemini.interactionsStream fatal', {
362
+ error,
363
+ source: 'gemini.interactionsStream',
364
+ })
365
+ yield {
366
+ type: EventType.RUN_ERROR,
367
+ model,
368
+ timestamp: Date.now(),
369
+ message:
370
+ error instanceof Error
371
+ ? error.message
372
+ : 'An unknown error occurred during the chat stream.',
373
+ ...(rawEvent !== undefined && { rawEvent }),
374
+ error: {
375
+ message:
376
+ error instanceof Error
377
+ ? error.message
378
+ : 'An unknown error occurred during the chat stream.',
379
+ },
380
+ }
381
+ }
382
+ }
383
+
164
384
  /**
165
385
  * Generate structured output using Gemini's native JSON response format.
166
386
  * Uses responseMimeType: 'application/json' and responseSchema for structured output.
@@ -595,29 +815,42 @@ export class GeminiTextAdapter<
595
815
  case 'audio':
596
816
  case 'video':
597
817
  case 'document': {
598
- if (part.source.type === 'data') {
599
- return {
600
- inlineData: {
601
- data: part.source.value,
602
- mimeType: part.source.mimeType,
603
- },
818
+ const geminiPart: Part =
819
+ part.source.type === 'data'
820
+ ? {
821
+ inlineData: {
822
+ data: part.source.value,
823
+ mimeType: part.source.mimeType,
824
+ },
825
+ }
826
+ : {
827
+ fileData: {
828
+ fileUri: part.source.value,
829
+ // For URL sources, use provided mimeType or fall back to
830
+ // reasonable defaults.
831
+ mimeType:
832
+ part.source.mimeType ?? DEFAULT_MEDIA_MIME_TYPES[part.type],
833
+ },
834
+ }
835
+
836
+ // Apply single-pass video sampling controls (fps / clip offsets) from
837
+ // the part metadata. `processing: 'agentic'` is handled separately via
838
+ // the Interactions API and never reaches this generateContent path.
839
+ if (part.type === 'video') {
840
+ const meta = part.metadata as GeminiVideoMetadata | undefined
841
+ const videoMetadata: VideoMetadata = {
842
+ ...(meta?.fps !== undefined && { fps: meta.fps }),
843
+ ...(meta?.startOffset !== undefined && {
844
+ startOffset: meta.startOffset,
845
+ }),
846
+ ...(meta?.endOffset !== undefined && { endOffset: meta.endOffset }),
604
847
  }
605
- } else {
606
- // For URL sources, use provided mimeType or fall back to reasonable defaults
607
- const defaultMimeType = {
608
- image: 'image/jpeg',
609
- audio: 'audio/mp3',
610
- video: 'video/mp4',
611
- document: 'application/pdf',
612
- }[part.type]
613
-
614
- return {
615
- fileData: {
616
- fileUri: part.source.value,
617
- mimeType: part.source.mimeType ?? defaultMimeType,
618
- },
848
+ if (Object.keys(videoMetadata).length > 0) {
849
+ geminiPart.videoMetadata = videoMetadata
619
850
  }
620
851
  }
852
+
853
+ return geminiPart
621
854
  }
622
855
  default: {
623
856
  const _exhaustiveCheck: never = part
@@ -9,6 +9,7 @@ import { createGeminiClient, getGeminiApiKeyFromEnv } from '../utils'
9
9
  import {
10
10
  getGeminiVideoDurationOptions,
11
11
  isInteractionsVideoModel,
12
+ parseGeminiOmniVideoSize,
12
13
  } from '../video/video-provider-options'
13
14
  import type { DurationOptions } from '@tanstack/ai/adapters'
14
15
  import type {
@@ -36,7 +37,6 @@ import type {
36
37
  GeminiVideoModelProviderOptionsByName,
37
38
  GeminiVideoModelSizeByName,
38
39
  GeminiVideoProviderOptions,
39
- GeminiVideoSize,
40
40
  } from '../video/video-provider-options'
41
41
  import type { GeminiClientConfig } from '../utils/client'
42
42
 
@@ -228,15 +228,19 @@ function interactionUsageToTokenUsage(
228
228
  * requires the API key (`x-goog-api-key` header or `?key=` query
229
229
  * parameter) to download.
230
230
  *
231
- * **Gemini Omni Flash** (`gemini-omni-flash-preview`) only serves the
232
- * Interactions API: `createVideoJob` creates a background interaction with
231
+ * **Gemini Omni Flash** (`gemini-omni-1.1-flash`, plus the deprecated
232
+ * `gemini-omni-flash-preview` alias) only serves the Interactions API:
233
+ * `createVideoJob` creates a background interaction with
233
234
  * `response_modalities: ['video']`, `getVideoStatus` polls it by id, and
234
235
  * `getVideoUrl` returns the inline base64 MP4 as a `data:` URL (or the
235
236
  * Files API URI when the server delivers by reference). Image and video
236
237
  * prompt parts are sent as interaction content blocks, grouped as images,
237
238
  * then videos, then the text prompt (interleaving is not preserved); pass
238
239
  * `modelOptions.previous_interaction_id` to conversationally edit a prior
239
- * Omni generation.
240
+ * Omni generation. `size` is an `aspectRatio_resolution` template
241
+ * (`'16:9'` or `'16:9_1080p'`); the optional suffix maps onto
242
+ * `response_format.resolution` (`'360p' | '720p' | '1080p' | '4k'`,
243
+ * default 720p).
240
244
  *
241
245
  * @experimental Video generation is an experimental feature and may change.
242
246
  */
@@ -264,7 +268,7 @@ export class GeminiVideoAdapter<
264
268
  async createVideoJob(
265
269
  options: VideoGenerationOptions<
266
270
  GeminiVideoModelProviderOptionsByName[TModel],
267
- GeminiVideoSize,
271
+ GeminiVideoModelSizeByName[TModel],
268
272
  GeminiVideoModelDurationByName[TModel]
269
273
  >,
270
274
  ): Promise<VideoJobResult> {
@@ -339,7 +343,7 @@ export class GeminiVideoAdapter<
339
343
  private async createInteractionsVideoJob(
340
344
  options: VideoGenerationOptions<
341
345
  GeminiVideoModelProviderOptionsByName[TModel],
342
- GeminiVideoSize,
346
+ GeminiVideoModelSizeByName[TModel],
343
347
  GeminiVideoModelDurationByName[TModel]
344
348
  >,
345
349
  ): Promise<VideoJobResult> {
@@ -384,16 +388,23 @@ export class GeminiVideoAdapter<
384
388
  )
385
389
  }
386
390
 
387
- // Aspect ratio and clip length ride on `response_format`. Duration is
388
- // a `"<seconds>s"` string, accepted anywhere in the 3–10s range
389
- // (fractional included) and defaulting to 10s when omitted — verified
390
- // against the live API; the docs don't publish the range constraints.
391
+ // Aspect ratio, clip length, and resolution ride on `response_format`.
392
+ // Duration is a `"<seconds>s"` string. Resolution comes from the
393
+ // optional `size` suffix (`'16:9_1080p'`), defaulting to 720p when
394
+ // omitted — https://ai.google.dev/gemini-api/docs/omni
395
+ const parsedSize =
396
+ size !== undefined ? parseGeminiOmniVideoSize(size) : undefined
391
397
  const responseFormat =
392
- size !== undefined || duration !== undefined
398
+ parsedSize !== undefined || duration !== undefined
393
399
  ? {
394
400
  response_format: {
395
401
  type: 'video' as const,
396
- ...(size !== undefined && { aspect_ratio: size }),
402
+ ...(parsedSize !== undefined && {
403
+ aspect_ratio: parsedSize.aspectRatio,
404
+ ...(parsedSize.resolution !== undefined && {
405
+ resolution: parsedSize.resolution,
406
+ }),
407
+ }),
397
408
  ...(duration !== undefined && { duration: `${duration}s` }),
398
409
  },
399
410
  }
@@ -659,6 +670,12 @@ export class GeminiVideoAdapter<
659
670
  }
660
671
  }
661
672
 
673
+ /** @deprecated Shuts down 2026-09-30. Use `gemini-omni-1.1-flash`. */
674
+ export function createGeminiVideo(
675
+ model: 'gemini-omni-flash-preview',
676
+ apiKey: string,
677
+ config?: Omit<GeminiVideoConfig, 'apiKey'>,
678
+ ): GeminiVideoAdapter<'gemini-omni-flash-preview'>
662
679
  /**
663
680
  * Creates a Gemini video adapter with an explicit API key.
664
681
  * Type resolution happens here at the call site.
@@ -681,6 +698,11 @@ export class GeminiVideoAdapter<
681
698
  * });
682
699
  * ```
683
700
  */
701
+ export function createGeminiVideo<TModel extends GeminiVideoModel>(
702
+ model: TModel,
703
+ apiKey: string,
704
+ config?: Omit<GeminiVideoConfig, 'apiKey'>,
705
+ ): GeminiVideoAdapter<TModel>
684
706
  export function createGeminiVideo<TModel extends GeminiVideoModel>(
685
707
  model: TModel,
686
708
  apiKey: string,
@@ -689,6 +711,11 @@ export function createGeminiVideo<TModel extends GeminiVideoModel>(
689
711
  return new GeminiVideoAdapter({ apiKey, ...config }, model)
690
712
  }
691
713
 
714
+ /** @deprecated Shuts down 2026-09-30. Use `gemini-omni-1.1-flash`. */
715
+ export function geminiVideo(
716
+ model: 'gemini-omni-flash-preview',
717
+ config?: Omit<GeminiVideoConfig, 'apiKey'>,
718
+ ): GeminiVideoAdapter<'gemini-omni-flash-preview'>
692
719
  /**
693
720
  * Creates a Gemini video adapter with automatic API key detection from environment variables.
694
721
  * Type resolution happens here at the call site.
@@ -719,6 +746,10 @@ export function createGeminiVideo<TModel extends GeminiVideoModel>(
719
746
  * const status = await getVideoJobStatus({ adapter, jobId });
720
747
  * ```
721
748
  */
749
+ export function geminiVideo<TModel extends GeminiVideoModel>(
750
+ model: TModel,
751
+ config?: Omit<GeminiVideoConfig, 'apiKey'>,
752
+ ): GeminiVideoAdapter<TModel>
722
753
  export function geminiVideo<TModel extends GeminiVideoModel>(
723
754
  model: TModel,
724
755
  config?: Omit<GeminiVideoConfig, 'apiKey'>,
@@ -0,0 +1,114 @@
1
+ import { FileState } from '@google/genai'
2
+ import { createGeminiClient, getGeminiApiKeyFromEnv } from '../utils'
3
+ import type { GeminiVideoMetadata } from '../message-types'
4
+ import type { VideoPart } from '@tanstack/ai'
5
+
6
+ /**
7
+ * A file uploaded to the Gemini Files API and ready to reference in a message.
8
+ */
9
+ export interface GeminiUploadedFile {
10
+ /** Resource name, e.g. `"files/abc123"`. */
11
+ name: string
12
+ /** File URI to reference from message content (as a `url` source). */
13
+ uri: string
14
+ /** MIME type reported by the Files API. */
15
+ mimeType: string
16
+ }
17
+
18
+ /**
19
+ * Options for {@link uploadGeminiFile}.
20
+ */
21
+ export interface GeminiUploadFileOptions {
22
+ /**
23
+ * API key. Falls back to `GOOGLE_API_KEY` / `GEMINI_API_KEY` from the
24
+ * environment when omitted.
25
+ */
26
+ apiKey?: string
27
+ /**
28
+ * MIME type of the file (e.g. `"video/mp4"`). Recommended so the Files API
29
+ * processes and serves the file with the correct type.
30
+ */
31
+ mimeType?: string
32
+ /** Poll interval while the file is `PROCESSING`, in ms. Default `5000`. */
33
+ pollIntervalMs?: number
34
+ /** Max time to wait for processing, in ms. Default `300000` (5 min). */
35
+ timeoutMs?: number
36
+ }
37
+
38
+ /**
39
+ * Upload a file via the Gemini Files API and wait until it is `ACTIVE`.
40
+ *
41
+ * Large media (notably video) must be uploaded rather than inlined as base64;
42
+ * the Files API processes uploads asynchronously. This wraps the upload +
43
+ * poll-until-ready loop and returns a reference you can drop into message
44
+ * content as a `url` source (see {@link geminiVideoPart}).
45
+ *
46
+ * @throws if the upload has no URI, or processing fails or times out.
47
+ */
48
+ export async function uploadGeminiFile(
49
+ file: string | Blob,
50
+ options: GeminiUploadFileOptions = {},
51
+ ): Promise<GeminiUploadedFile> {
52
+ const {
53
+ apiKey = getGeminiApiKeyFromEnv(),
54
+ mimeType,
55
+ pollIntervalMs = 5000,
56
+ timeoutMs = 300_000,
57
+ } = options
58
+
59
+ const client = createGeminiClient({ apiKey })
60
+
61
+ let uploaded = await client.files.upload({
62
+ file,
63
+ ...(mimeType && { config: { mimeType } }),
64
+ })
65
+
66
+ const fileName = uploaded.name
67
+ if (!fileName) {
68
+ throw new Error('Gemini file upload did not return a file name.')
69
+ }
70
+
71
+ const deadline = Date.now() + timeoutMs
72
+ while (uploaded.state === FileState.PROCESSING) {
73
+ if (Date.now() > deadline) {
74
+ throw new Error(
75
+ `Gemini file processing timed out after ${timeoutMs}ms (${fileName}).`,
76
+ )
77
+ }
78
+ await new Promise((resolve) => setTimeout(resolve, pollIntervalMs))
79
+ uploaded = await client.files.get({ name: fileName })
80
+ }
81
+
82
+ if (uploaded.state === FileState.FAILED) {
83
+ throw new Error(
84
+ `Gemini file processing failed: ${uploaded.error?.message ?? String(uploaded.state)}`,
85
+ )
86
+ }
87
+ if (!uploaded.uri) {
88
+ throw new Error('Gemini file upload did not return a URI.')
89
+ }
90
+
91
+ return {
92
+ name: fileName,
93
+ uri: uploaded.uri,
94
+ mimeType: uploaded.mimeType ?? mimeType ?? 'application/octet-stream',
95
+ }
96
+ }
97
+
98
+ /**
99
+ * Build a TanStack AI video content part from an uploaded Gemini file.
100
+ *
101
+ * Pass `metadata` to control understanding — e.g.
102
+ * `{ processing: 'agentic' }` to route through the agentic Interactions path,
103
+ * or `{ fps, startOffset, endOffset }` for single-pass sampling controls.
104
+ */
105
+ export function geminiVideoPart(
106
+ file: GeminiUploadedFile,
107
+ metadata?: GeminiVideoMetadata,
108
+ ): VideoPart<GeminiVideoMetadata> {
109
+ return {
110
+ type: 'video',
111
+ source: { type: 'url', value: file.uri, mimeType: file.mimeType },
112
+ ...(metadata && { metadata }),
113
+ }
114
+ }
package/src/index.ts CHANGED
@@ -60,6 +60,14 @@ export type {
60
60
  // having to add `@google/genai` to their own dependencies.
61
61
  export { HarmBlockThreshold, HarmCategory } from '@google/genai'
62
62
 
63
+ // Files API helpers — upload + poll a file (e.g. video) until it is ACTIVE
64
+ export {
65
+ uploadGeminiFile,
66
+ geminiVideoPart,
67
+ type GeminiUploadedFile,
68
+ type GeminiUploadFileOptions,
69
+ } from './files/index'
70
+
63
71
  // Embedding adapter - for embedding vectors
64
72
  export {
65
73
  GeminiEmbeddingAdapter,
@@ -108,10 +116,13 @@ export {
108
116
  GEMINI_VIDEO_DURATIONS,
109
117
  getGeminiVideoDurationOptions,
110
118
  isInteractionsVideoModel,
119
+ parseGeminiOmniVideoSize,
111
120
  } from './video/video-provider-options'
112
121
  export type {
113
122
  GeminiInteractionsVideoModel,
114
123
  GeminiOmniVideoProviderOptions,
124
+ GeminiOmniVideoResolution,
125
+ GeminiOmniVideoSize,
115
126
  GeminiVideoModel,
116
127
  GeminiVideoModelDurationByName,
117
128
  GeminiVideoModelInputModalitiesByName,
@@ -165,6 +176,7 @@ export type {
165
176
  GeminiImageMetadata,
166
177
  GeminiAudioMetadata,
167
178
  GeminiVideoMetadata,
179
+ GeminiVideoProcessing,
168
180
  GeminiDocumentMetadata,
169
181
  GeminiMessageMetadataByModality,
170
182
  } from './message-types'
@@ -87,6 +87,19 @@ export interface GeminiAudioMetadata {
87
87
  mimeType?: GeminiAudioMimeType
88
88
  }
89
89
 
90
+ /**
91
+ * How Gemini processes a video for understanding.
92
+ *
93
+ * - `static` (default): single-pass frame sampling via `generateContent`.
94
+ * - `agentic`: multi-pass "agentic" video understanding, GA on the
95
+ * `agentic_video`-capable flash models (`gemini-3.7-flash`,
96
+ * `gemini-3.6-flash`, `gemini-3.5-flash-lite`). The text adapter routes the
97
+ * request through the Interactions API instead of `generateContent`. With
98
+ * `agentic`, the sampling rate is expressed in the text prompt (e.g. "watch
99
+ * it at 0.5 fps"), not via `fps`.
100
+ */
101
+ export type GeminiVideoProcessing = 'agentic' | 'static'
102
+
90
103
  /**
91
104
  * Metadata for Gemini video content parts.
92
105
  */
@@ -98,6 +111,30 @@ export interface GeminiVideoMetadata {
98
111
  * @see https://ai.google.dev/gemini-api/docs/vision#video-requirements
99
112
  */
100
113
  mimeType?: GeminiVideoMimeType
114
+ /**
115
+ * How the model processes this video for understanding. When set, the
116
+ * adapter routes the request through the Gemini Interactions API. Omit for
117
+ * the default single-pass `generateContent` behavior.
118
+ */
119
+ processing?: GeminiVideoProcessing
120
+ /**
121
+ * Frame-rate sampling density (frames per second) for single-pass
122
+ * (`generateContent`) understanding. Valid range (0, 24]; defaults to 1.0
123
+ * on the server. Ignored when `processing` is `agentic`.
124
+ *
125
+ * @see https://ai.google.dev/gemini-api/docs/vision#customize-frame-rate
126
+ */
127
+ fps?: number
128
+ /**
129
+ * Clip start offset, as a decimal number of seconds with an `s` suffix
130
+ * (e.g. `"10.5s"`). Restricts understanding to a segment of the video.
131
+ */
132
+ startOffset?: string
133
+ /**
134
+ * Clip end offset, as a decimal number of seconds with an `s` suffix
135
+ * (e.g. `"45s"`). Restricts understanding to a segment of the video.
136
+ */
137
+ endOffset?: string
101
138
  }
102
139
 
103
140
  /**