@tanstack/ai-gemini 0.26.5 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -28,6 +28,7 @@ import type {
28
28
  GoogleGenAI,
29
29
  Part,
30
30
  ThinkingLevel,
31
+ VideoMetadata,
31
32
  } from '@google/genai'
32
33
  import type {
33
34
  ContentPart,
@@ -40,9 +41,113 @@ import type { ExternalTextProviderOptions } from '../text/text-provider-options'
40
41
  import type {
41
42
  GeminiMessageMetadataByModality,
42
43
  GeminiToolCallMetadata,
44
+ GeminiVideoMetadata,
45
+ GeminiVideoProcessing,
43
46
  } from '../message-types'
44
47
  import type { GeminiClientConfig } from '../utils/client'
45
48
 
49
+ /**
50
+ * Fallback MIME types for URL-sourced media parts that don't specify one.
51
+ */
52
+ const DEFAULT_MEDIA_MIME_TYPES = {
53
+ image: 'image/jpeg',
54
+ audio: 'audio/mp3',
55
+ video: 'video/mp4',
56
+ document: 'application/pdf',
57
+ } as const
58
+
59
+ /**
60
+ * Content block shape for an Interactions API `input` step. The installed
61
+ * @google/genai types predate the video `processing` field, so we model the
62
+ * subset we emit and cast at the call site.
63
+ */
64
+ type InteractionContent =
65
+ | { type: 'text'; text: string }
66
+ | {
67
+ type: 'video'
68
+ uri?: string
69
+ data?: string
70
+ mime_type?: string
71
+ processing?: GeminiVideoProcessing
72
+ }
73
+ | {
74
+ type: 'image' | 'audio' | 'document'
75
+ uri?: string
76
+ data?: string
77
+ mime_type?: string
78
+ }
79
+
80
+ interface InteractionStep {
81
+ type: 'user_input' | 'model_output'
82
+ content: Array<InteractionContent>
83
+ }
84
+
85
+ /** True when any message carries a video part requesting agentic processing. */
86
+ function hasAgenticVideo(messages: Array<ModelMessage>): boolean {
87
+ return messages.some(
88
+ (msg) =>
89
+ Array.isArray(msg.content) &&
90
+ msg.content.some(
91
+ (part) =>
92
+ part.type === 'video' &&
93
+ (part.metadata as GeminiVideoMetadata | undefined)?.processing ===
94
+ 'agentic',
95
+ ),
96
+ )
97
+ }
98
+
99
+ /** Convert a single content part to an Interactions API content block. */
100
+ function contentPartToInteraction(part: ContentPart): InteractionContent {
101
+ if (part.type === 'text') {
102
+ return { type: 'text', text: part.content }
103
+ }
104
+
105
+ const source = part.source
106
+ const mimeType =
107
+ source.type === 'data'
108
+ ? source.mimeType
109
+ : (source.mimeType ?? DEFAULT_MEDIA_MIME_TYPES[part.type])
110
+ const base =
111
+ source.type === 'data'
112
+ ? { data: source.value, mime_type: mimeType }
113
+ : { uri: source.value, mime_type: mimeType }
114
+
115
+ if (part.type === 'video') {
116
+ const processing = (part.metadata as GeminiVideoMetadata | undefined)
117
+ ?.processing
118
+ return { type: 'video', ...base, ...(processing && { processing }) }
119
+ }
120
+ return { type: part.type, ...base }
121
+ }
122
+
123
+ /**
124
+ * Build the Interactions API `input` from chat messages. Each user/assistant
125
+ * message becomes a `user_input` / `model_output` step wrapping its content
126
+ * blocks — the wrapping the Python SDK performs implicitly but the JS SDK
127
+ * does not. Tool messages are skipped (unsupported on this path).
128
+ */
129
+ function buildInteractionsInput(
130
+ messages: Array<ModelMessage>,
131
+ ): Array<InteractionStep> {
132
+ const steps: Array<InteractionStep> = []
133
+ for (const msg of messages) {
134
+ if (msg.role === 'tool') continue
135
+ const stepType = msg.role === 'assistant' ? 'model_output' : 'user_input'
136
+ const content: Array<InteractionContent> = []
137
+ if (Array.isArray(msg.content)) {
138
+ for (const part of msg.content) {
139
+ content.push(contentPartToInteraction(part))
140
+ }
141
+ } else if (msg.content) {
142
+ content.push({ type: 'text', text: msg.content })
143
+ }
144
+ if (content.length > 0) {
145
+ steps.push({ type: stepType, content })
146
+ }
147
+ }
148
+ return steps
149
+ }
150
+
46
151
  /**
47
152
  * Configuration for Gemini text adapter
48
153
  */
@@ -122,6 +227,13 @@ export class GeminiTextAdapter<
122
227
  async *chatStream(
123
228
  options: TextOptions<GeminiTextProviderOptions>,
124
229
  ): AsyncIterable<AdapterYieldChunk> {
230
+ // Agentic video understanding is only exposed through the Interactions API,
231
+ // not generateContent. Detect it and take that path instead.
232
+ if (hasAgenticVideo(options.messages)) {
233
+ yield* this.interactionsStream(options)
234
+ return
235
+ }
236
+
125
237
  const mappedOptions = this.mapCommonOptionsToGemini(options)
126
238
  const { logger } = options
127
239
 
@@ -161,6 +273,114 @@ export class GeminiTextAdapter<
161
273
  }
162
274
  }
163
275
 
276
+ /**
277
+ * Agentic video-understanding path via the Interactions API.
278
+ *
279
+ * The Interactions API (unlike `generateContent`) requires message parts to
280
+ * be wrapped in `user_input` / `model_output` steps, and it accepts the
281
+ * `processing: 'agentic'` video flag. This is a non-streaming call whose
282
+ * single text result is re-emitted as AG-UI stream chunks.
283
+ */
284
+ private async *interactionsStream(
285
+ options: TextOptions<GeminiTextProviderOptions>,
286
+ ): AsyncIterable<AdapterYieldChunk> {
287
+ const model = options.model
288
+ const { logger } = options
289
+ const runId = options.runId ?? generateId(this.name)
290
+ const threadId = options.threadId ?? generateId(this.name)
291
+ const messageId = generateId(this.name)
292
+
293
+ try {
294
+ logger.request(
295
+ `activity=chat provider=gemini model=${model} messages=${options.messages.length} mode=interactions-agentic-video`,
296
+ { provider: 'gemini', model },
297
+ )
298
+
299
+ const normalizedPrompts = normalizeSystemPrompts(options.systemPrompts)
300
+ const systemInstruction =
301
+ normalizedPrompts.length > 0
302
+ ? normalizedPrompts.map((p) => p.content).join('\n')
303
+ : undefined
304
+
305
+ const input = buildInteractionsInput(options.messages)
306
+
307
+ // The installed @google/genai (2.10.0) Interactions `VideoContent` type
308
+ // predates the `processing` field, so the structurally-built input is
309
+ // cast at the call boundary. The SDK forwards it to the wire unchanged.
310
+ const interaction = await this.client.interactions.create({
311
+ model,
312
+ ...(systemInstruction !== undefined && {
313
+ system_instruction: systemInstruction,
314
+ }),
315
+ input: input as never,
316
+ })
317
+
318
+ const text = interaction.output_text ?? ''
319
+
320
+ yield {
321
+ type: EventType.RUN_STARTED,
322
+ runId,
323
+ threadId,
324
+ model,
325
+ timestamp: Date.now(),
326
+ parentRunId: options.parentRunId,
327
+ }
328
+ yield {
329
+ type: EventType.TEXT_MESSAGE_START,
330
+ messageId,
331
+ model,
332
+ timestamp: Date.now(),
333
+ role: 'assistant',
334
+ }
335
+ if (text) {
336
+ yield {
337
+ type: EventType.TEXT_MESSAGE_CONTENT,
338
+ messageId,
339
+ model,
340
+ timestamp: Date.now(),
341
+ delta: text,
342
+ content: text,
343
+ }
344
+ }
345
+ yield {
346
+ type: EventType.TEXT_MESSAGE_END,
347
+ messageId,
348
+ model,
349
+ timestamp: Date.now(),
350
+ }
351
+ yield {
352
+ type: EventType.RUN_FINISHED,
353
+ runId,
354
+ threadId,
355
+ model,
356
+ timestamp: Date.now(),
357
+ finishReason: 'stop',
358
+ }
359
+ } catch (error) {
360
+ const rawEvent = toRunErrorRawEvent(error)
361
+ logger.errors('gemini.interactionsStream fatal', {
362
+ error,
363
+ source: 'gemini.interactionsStream',
364
+ })
365
+ yield {
366
+ type: EventType.RUN_ERROR,
367
+ model,
368
+ timestamp: Date.now(),
369
+ message:
370
+ error instanceof Error
371
+ ? error.message
372
+ : 'An unknown error occurred during the chat stream.',
373
+ ...(rawEvent !== undefined && { rawEvent }),
374
+ error: {
375
+ message:
376
+ error instanceof Error
377
+ ? error.message
378
+ : 'An unknown error occurred during the chat stream.',
379
+ },
380
+ }
381
+ }
382
+ }
383
+
164
384
  /**
165
385
  * Generate structured output using Gemini's native JSON response format.
166
386
  * Uses responseMimeType: 'application/json' and responseSchema for structured output.
@@ -595,29 +815,42 @@ export class GeminiTextAdapter<
595
815
  case 'audio':
596
816
  case 'video':
597
817
  case 'document': {
598
- if (part.source.type === 'data') {
599
- return {
600
- inlineData: {
601
- data: part.source.value,
602
- mimeType: part.source.mimeType,
603
- },
818
+ const geminiPart: Part =
819
+ part.source.type === 'data'
820
+ ? {
821
+ inlineData: {
822
+ data: part.source.value,
823
+ mimeType: part.source.mimeType,
824
+ },
825
+ }
826
+ : {
827
+ fileData: {
828
+ fileUri: part.source.value,
829
+ // For URL sources, use provided mimeType or fall back to
830
+ // reasonable defaults.
831
+ mimeType:
832
+ part.source.mimeType ?? DEFAULT_MEDIA_MIME_TYPES[part.type],
833
+ },
834
+ }
835
+
836
+ // Apply single-pass video sampling controls (fps / clip offsets) from
837
+ // the part metadata. `processing: 'agentic'` is handled separately via
838
+ // the Interactions API and never reaches this generateContent path.
839
+ if (part.type === 'video') {
840
+ const meta = part.metadata as GeminiVideoMetadata | undefined
841
+ const videoMetadata: VideoMetadata = {
842
+ ...(meta?.fps !== undefined && { fps: meta.fps }),
843
+ ...(meta?.startOffset !== undefined && {
844
+ startOffset: meta.startOffset,
845
+ }),
846
+ ...(meta?.endOffset !== undefined && { endOffset: meta.endOffset }),
604
847
  }
605
- } else {
606
- // For URL sources, use provided mimeType or fall back to reasonable defaults
607
- const defaultMimeType = {
608
- image: 'image/jpeg',
609
- audio: 'audio/mp3',
610
- video: 'video/mp4',
611
- document: 'application/pdf',
612
- }[part.type]
613
-
614
- return {
615
- fileData: {
616
- fileUri: part.source.value,
617
- mimeType: part.source.mimeType ?? defaultMimeType,
618
- },
848
+ if (Object.keys(videoMetadata).length > 0) {
849
+ geminiPart.videoMetadata = videoMetadata
619
850
  }
620
851
  }
852
+
853
+ return geminiPart
621
854
  }
622
855
  default: {
623
856
  const _exhaustiveCheck: never = part
@@ -0,0 +1,114 @@
1
+ import { FileState } from '@google/genai'
2
+ import { createGeminiClient, getGeminiApiKeyFromEnv } from '../utils'
3
+ import type { GeminiVideoMetadata } from '../message-types'
4
+ import type { VideoPart } from '@tanstack/ai'
5
+
6
+ /**
7
+ * A file uploaded to the Gemini Files API and ready to reference in a message.
8
+ */
9
+ export interface GeminiUploadedFile {
10
+ /** Resource name, e.g. `"files/abc123"`. */
11
+ name: string
12
+ /** File URI to reference from message content (as a `url` source). */
13
+ uri: string
14
+ /** MIME type reported by the Files API. */
15
+ mimeType: string
16
+ }
17
+
18
+ /**
19
+ * Options for {@link uploadGeminiFile}.
20
+ */
21
+ export interface GeminiUploadFileOptions {
22
+ /**
23
+ * API key. Falls back to `GOOGLE_API_KEY` / `GEMINI_API_KEY` from the
24
+ * environment when omitted.
25
+ */
26
+ apiKey?: string
27
+ /**
28
+ * MIME type of the file (e.g. `"video/mp4"`). Recommended so the Files API
29
+ * processes and serves the file with the correct type.
30
+ */
31
+ mimeType?: string
32
+ /** Poll interval while the file is `PROCESSING`, in ms. Default `5000`. */
33
+ pollIntervalMs?: number
34
+ /** Max time to wait for processing, in ms. Default `300000` (5 min). */
35
+ timeoutMs?: number
36
+ }
37
+
38
+ /**
39
+ * Upload a file via the Gemini Files API and wait until it is `ACTIVE`.
40
+ *
41
+ * Large media (notably video) must be uploaded rather than inlined as base64;
42
+ * the Files API processes uploads asynchronously. This wraps the upload +
43
+ * poll-until-ready loop and returns a reference you can drop into message
44
+ * content as a `url` source (see {@link geminiVideoPart}).
45
+ *
46
+ * @throws if the upload has no URI, or processing fails or times out.
47
+ */
48
+ export async function uploadGeminiFile(
49
+ file: string | Blob,
50
+ options: GeminiUploadFileOptions = {},
51
+ ): Promise<GeminiUploadedFile> {
52
+ const {
53
+ apiKey = getGeminiApiKeyFromEnv(),
54
+ mimeType,
55
+ pollIntervalMs = 5000,
56
+ timeoutMs = 300_000,
57
+ } = options
58
+
59
+ const client = createGeminiClient({ apiKey })
60
+
61
+ let uploaded = await client.files.upload({
62
+ file,
63
+ ...(mimeType && { config: { mimeType } }),
64
+ })
65
+
66
+ const fileName = uploaded.name
67
+ if (!fileName) {
68
+ throw new Error('Gemini file upload did not return a file name.')
69
+ }
70
+
71
+ const deadline = Date.now() + timeoutMs
72
+ while (uploaded.state === FileState.PROCESSING) {
73
+ if (Date.now() > deadline) {
74
+ throw new Error(
75
+ `Gemini file processing timed out after ${timeoutMs}ms (${fileName}).`,
76
+ )
77
+ }
78
+ await new Promise((resolve) => setTimeout(resolve, pollIntervalMs))
79
+ uploaded = await client.files.get({ name: fileName })
80
+ }
81
+
82
+ if (uploaded.state === FileState.FAILED) {
83
+ throw new Error(
84
+ `Gemini file processing failed: ${uploaded.error?.message ?? String(uploaded.state)}`,
85
+ )
86
+ }
87
+ if (!uploaded.uri) {
88
+ throw new Error('Gemini file upload did not return a URI.')
89
+ }
90
+
91
+ return {
92
+ name: fileName,
93
+ uri: uploaded.uri,
94
+ mimeType: uploaded.mimeType ?? mimeType ?? 'application/octet-stream',
95
+ }
96
+ }
97
+
98
+ /**
99
+ * Build a TanStack AI video content part from an uploaded Gemini file.
100
+ *
101
+ * Pass `metadata` to control understanding — e.g.
102
+ * `{ processing: 'agentic' }` to route through the agentic Interactions path,
103
+ * or `{ fps, startOffset, endOffset }` for single-pass sampling controls.
104
+ */
105
+ export function geminiVideoPart(
106
+ file: GeminiUploadedFile,
107
+ metadata?: GeminiVideoMetadata,
108
+ ): VideoPart<GeminiVideoMetadata> {
109
+ return {
110
+ type: 'video',
111
+ source: { type: 'url', value: file.uri, mimeType: file.mimeType },
112
+ ...(metadata && { metadata }),
113
+ }
114
+ }
package/src/index.ts CHANGED
@@ -60,6 +60,14 @@ export type {
60
60
  // having to add `@google/genai` to their own dependencies.
61
61
  export { HarmBlockThreshold, HarmCategory } from '@google/genai'
62
62
 
63
+ // Files API helpers — upload + poll a file (e.g. video) until it is ACTIVE
64
+ export {
65
+ uploadGeminiFile,
66
+ geminiVideoPart,
67
+ type GeminiUploadedFile,
68
+ type GeminiUploadFileOptions,
69
+ } from './files/index'
70
+
63
71
  // Embedding adapter - for embedding vectors
64
72
  export {
65
73
  GeminiEmbeddingAdapter,
@@ -168,6 +176,7 @@ export type {
168
176
  GeminiImageMetadata,
169
177
  GeminiAudioMetadata,
170
178
  GeminiVideoMetadata,
179
+ GeminiVideoProcessing,
171
180
  GeminiDocumentMetadata,
172
181
  GeminiMessageMetadataByModality,
173
182
  } from './message-types'
@@ -87,6 +87,19 @@ export interface GeminiAudioMetadata {
87
87
  mimeType?: GeminiAudioMimeType
88
88
  }
89
89
 
90
+ /**
91
+ * How Gemini processes a video for understanding.
92
+ *
93
+ * - `static` (default): single-pass frame sampling via `generateContent`.
94
+ * - `agentic`: multi-pass "agentic" video understanding, GA on the
95
+ * `agentic_video`-capable flash models (`gemini-3.7-flash`,
96
+ * `gemini-3.6-flash`, `gemini-3.5-flash-lite`). The text adapter routes the
97
+ * request through the Interactions API instead of `generateContent`. With
98
+ * `agentic`, the sampling rate is expressed in the text prompt (e.g. "watch
99
+ * it at 0.5 fps"), not via `fps`.
100
+ */
101
+ export type GeminiVideoProcessing = 'agentic' | 'static'
102
+
90
103
  /**
91
104
  * Metadata for Gemini video content parts.
92
105
  */
@@ -98,6 +111,30 @@ export interface GeminiVideoMetadata {
98
111
  * @see https://ai.google.dev/gemini-api/docs/vision#video-requirements
99
112
  */
100
113
  mimeType?: GeminiVideoMimeType
114
+ /**
115
+ * How the model processes this video for understanding. When set, the
116
+ * adapter routes the request through the Gemini Interactions API. Omit for
117
+ * the default single-pass `generateContent` behavior.
118
+ */
119
+ processing?: GeminiVideoProcessing
120
+ /**
121
+ * Frame-rate sampling density (frames per second) for single-pass
122
+ * (`generateContent`) understanding. Valid range (0, 24]; defaults to 1.0
123
+ * on the server. Ignored when `processing` is `agentic`.
124
+ *
125
+ * @see https://ai.google.dev/gemini-api/docs/vision#customize-frame-rate
126
+ */
127
+ fps?: number
128
+ /**
129
+ * Clip start offset, as a decimal number of seconds with an `s` suffix
130
+ * (e.g. `"10.5s"`). Restricts understanding to a segment of the video.
131
+ */
132
+ startOffset?: string
133
+ /**
134
+ * Clip end offset, as a decimal number of seconds with an `s` suffix
135
+ * (e.g. `"45s"`). Restricts understanding to a segment of the video.
136
+ */
137
+ endOffset?: string
101
138
  }
102
139
 
103
140
  /**
package/src/model-meta.ts CHANGED
@@ -14,6 +14,7 @@ interface ModelMeta<TProviderOptions = unknown> {
14
14
  input: Array<'text' | 'image' | 'audio' | 'video' | 'document'>
15
15
  output: Array<'text' | 'image' | 'audio' | 'video'>
16
16
  capabilities?: Array<
17
+ | 'agentic_video'
17
18
  | 'audio_generation'
18
19
  | 'batch_api'
19
20
  | 'caching'
@@ -879,6 +880,7 @@ const GEMINI_3_7_FLASH = {
879
880
  input: ['text', 'image', 'video', 'audio', 'document'],
880
881
  output: ['text'],
881
882
  capabilities: [
883
+ 'agentic_video',
882
884
  'batch_api',
883
885
  'caching',
884
886
  'function_calling',
@@ -923,6 +925,7 @@ const GEMINI_3_6_FLASH = {
923
925
  input: ['text', 'image', 'video', 'audio', 'document'],
924
926
  output: ['text'],
925
927
  capabilities: [
928
+ 'agentic_video',
926
929
  'batch_api',
927
930
  'caching',
928
931
  'function_calling',
@@ -1006,6 +1009,7 @@ const GEMINI_3_5_FLASH_LITE = {
1006
1009
  input: ['text', 'image', 'video', 'audio', 'document'],
1007
1010
  output: ['text'],
1008
1011
  capabilities: [
1012
+ 'agentic_video',
1009
1013
  'batch_api',
1010
1014
  'caching',
1011
1015
  'function_calling',