@tanstack/ai-gemini 0.26.5 → 0.28.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -9
- package/dist/esm/adapters/text.d.ts +9 -0
- package/dist/esm/adapters/text.js +165 -13
- package/dist/esm/adapters/text.js.map +1 -1
- package/dist/esm/experimental/text-interactions/adapter.d.ts +20 -12
- package/dist/esm/experimental/text-interactions/adapter.js.map +1 -1
- package/dist/esm/files/index.d.ts +51 -0
- package/dist/esm/files/index.js +59 -0
- package/dist/esm/files/index.js.map +1 -0
- package/dist/esm/index.d.ts +2 -1
- package/dist/esm/index.js +2 -1
- package/dist/esm/message-types.d.ts +36 -0
- package/dist/esm/model-meta.d.ts +28 -4
- package/dist/esm/model-meta.js +44 -0
- package/dist/esm/model-meta.js.map +1 -1
- package/dist/esm/text/text-provider-options.d.ts +2 -2
- package/package.json +3 -3
- package/src/adapters/text.ts +253 -20
- package/src/experimental/text-interactions/adapter.ts +28 -11
- package/src/files/index.ts +114 -0
- package/src/index.ts +9 -0
- package/src/message-types.ts +37 -0
- package/src/model-meta.ts +57 -0
- package/src/text/text-provider-options.ts +5 -2
package/src/adapters/text.ts
CHANGED
|
@@ -28,6 +28,7 @@ import type {
|
|
|
28
28
|
GoogleGenAI,
|
|
29
29
|
Part,
|
|
30
30
|
ThinkingLevel,
|
|
31
|
+
VideoMetadata,
|
|
31
32
|
} from '@google/genai'
|
|
32
33
|
import type {
|
|
33
34
|
ContentPart,
|
|
@@ -40,9 +41,113 @@ import type { ExternalTextProviderOptions } from '../text/text-provider-options'
|
|
|
40
41
|
import type {
|
|
41
42
|
GeminiMessageMetadataByModality,
|
|
42
43
|
GeminiToolCallMetadata,
|
|
44
|
+
GeminiVideoMetadata,
|
|
45
|
+
GeminiVideoProcessing,
|
|
43
46
|
} from '../message-types'
|
|
44
47
|
import type { GeminiClientConfig } from '../utils/client'
|
|
45
48
|
|
|
49
|
+
/**
|
|
50
|
+
* Fallback MIME types for URL-sourced media parts that don't specify one.
|
|
51
|
+
*/
|
|
52
|
+
const DEFAULT_MEDIA_MIME_TYPES = {
|
|
53
|
+
image: 'image/jpeg',
|
|
54
|
+
audio: 'audio/mp3',
|
|
55
|
+
video: 'video/mp4',
|
|
56
|
+
document: 'application/pdf',
|
|
57
|
+
} as const
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* Content block shape for an Interactions API `input` step. The installed
|
|
61
|
+
* @google/genai types predate the video `processing` field, so we model the
|
|
62
|
+
* subset we emit and cast at the call site.
|
|
63
|
+
*/
|
|
64
|
+
type InteractionContent =
|
|
65
|
+
| { type: 'text'; text: string }
|
|
66
|
+
| {
|
|
67
|
+
type: 'video'
|
|
68
|
+
uri?: string
|
|
69
|
+
data?: string
|
|
70
|
+
mime_type?: string
|
|
71
|
+
processing?: GeminiVideoProcessing
|
|
72
|
+
}
|
|
73
|
+
| {
|
|
74
|
+
type: 'image' | 'audio' | 'document'
|
|
75
|
+
uri?: string
|
|
76
|
+
data?: string
|
|
77
|
+
mime_type?: string
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
interface InteractionStep {
|
|
81
|
+
type: 'user_input' | 'model_output'
|
|
82
|
+
content: Array<InteractionContent>
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/** True when any message carries a video part requesting agentic processing. */
|
|
86
|
+
function hasAgenticVideo(messages: Array<ModelMessage>): boolean {
|
|
87
|
+
return messages.some(
|
|
88
|
+
(msg) =>
|
|
89
|
+
Array.isArray(msg.content) &&
|
|
90
|
+
msg.content.some(
|
|
91
|
+
(part) =>
|
|
92
|
+
part.type === 'video' &&
|
|
93
|
+
(part.metadata as GeminiVideoMetadata | undefined)?.processing ===
|
|
94
|
+
'agentic',
|
|
95
|
+
),
|
|
96
|
+
)
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/** Convert a single content part to an Interactions API content block. */
|
|
100
|
+
function contentPartToInteraction(part: ContentPart): InteractionContent {
|
|
101
|
+
if (part.type === 'text') {
|
|
102
|
+
return { type: 'text', text: part.content }
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
const source = part.source
|
|
106
|
+
const mimeType =
|
|
107
|
+
source.type === 'data'
|
|
108
|
+
? source.mimeType
|
|
109
|
+
: (source.mimeType ?? DEFAULT_MEDIA_MIME_TYPES[part.type])
|
|
110
|
+
const base =
|
|
111
|
+
source.type === 'data'
|
|
112
|
+
? { data: source.value, mime_type: mimeType }
|
|
113
|
+
: { uri: source.value, mime_type: mimeType }
|
|
114
|
+
|
|
115
|
+
if (part.type === 'video') {
|
|
116
|
+
const processing = (part.metadata as GeminiVideoMetadata | undefined)
|
|
117
|
+
?.processing
|
|
118
|
+
return { type: 'video', ...base, ...(processing && { processing }) }
|
|
119
|
+
}
|
|
120
|
+
return { type: part.type, ...base }
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Build the Interactions API `input` from chat messages. Each user/assistant
|
|
125
|
+
* message becomes a `user_input` / `model_output` step wrapping its content
|
|
126
|
+
* blocks — the wrapping the Python SDK performs implicitly but the JS SDK
|
|
127
|
+
* does not. Tool messages are skipped (unsupported on this path).
|
|
128
|
+
*/
|
|
129
|
+
function buildInteractionsInput(
|
|
130
|
+
messages: Array<ModelMessage>,
|
|
131
|
+
): Array<InteractionStep> {
|
|
132
|
+
const steps: Array<InteractionStep> = []
|
|
133
|
+
for (const msg of messages) {
|
|
134
|
+
if (msg.role === 'tool') continue
|
|
135
|
+
const stepType = msg.role === 'assistant' ? 'model_output' : 'user_input'
|
|
136
|
+
const content: Array<InteractionContent> = []
|
|
137
|
+
if (Array.isArray(msg.content)) {
|
|
138
|
+
for (const part of msg.content) {
|
|
139
|
+
content.push(contentPartToInteraction(part))
|
|
140
|
+
}
|
|
141
|
+
} else if (msg.content) {
|
|
142
|
+
content.push({ type: 'text', text: msg.content })
|
|
143
|
+
}
|
|
144
|
+
if (content.length > 0) {
|
|
145
|
+
steps.push({ type: stepType, content })
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
return steps
|
|
149
|
+
}
|
|
150
|
+
|
|
46
151
|
/**
|
|
47
152
|
* Configuration for Gemini text adapter
|
|
48
153
|
*/
|
|
@@ -122,6 +227,13 @@ export class GeminiTextAdapter<
|
|
|
122
227
|
async *chatStream(
|
|
123
228
|
options: TextOptions<GeminiTextProviderOptions>,
|
|
124
229
|
): AsyncIterable<AdapterYieldChunk> {
|
|
230
|
+
// Agentic video understanding is only exposed through the Interactions API,
|
|
231
|
+
// not generateContent. Detect it and take that path instead.
|
|
232
|
+
if (hasAgenticVideo(options.messages)) {
|
|
233
|
+
yield* this.interactionsStream(options)
|
|
234
|
+
return
|
|
235
|
+
}
|
|
236
|
+
|
|
125
237
|
const mappedOptions = this.mapCommonOptionsToGemini(options)
|
|
126
238
|
const { logger } = options
|
|
127
239
|
|
|
@@ -161,6 +273,114 @@ export class GeminiTextAdapter<
|
|
|
161
273
|
}
|
|
162
274
|
}
|
|
163
275
|
|
|
276
|
+
/**
|
|
277
|
+
* Agentic video-understanding path via the Interactions API.
|
|
278
|
+
*
|
|
279
|
+
* The Interactions API (unlike `generateContent`) requires message parts to
|
|
280
|
+
* be wrapped in `user_input` / `model_output` steps, and it accepts the
|
|
281
|
+
* `processing: 'agentic'` video flag. This is a non-streaming call whose
|
|
282
|
+
* single text result is re-emitted as AG-UI stream chunks.
|
|
283
|
+
*/
|
|
284
|
+
private async *interactionsStream(
|
|
285
|
+
options: TextOptions<GeminiTextProviderOptions>,
|
|
286
|
+
): AsyncIterable<AdapterYieldChunk> {
|
|
287
|
+
const model = options.model
|
|
288
|
+
const { logger } = options
|
|
289
|
+
const runId = options.runId ?? generateId(this.name)
|
|
290
|
+
const threadId = options.threadId ?? generateId(this.name)
|
|
291
|
+
const messageId = generateId(this.name)
|
|
292
|
+
|
|
293
|
+
try {
|
|
294
|
+
logger.request(
|
|
295
|
+
`activity=chat provider=gemini model=${model} messages=${options.messages.length} mode=interactions-agentic-video`,
|
|
296
|
+
{ provider: 'gemini', model },
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
const normalizedPrompts = normalizeSystemPrompts(options.systemPrompts)
|
|
300
|
+
const systemInstruction =
|
|
301
|
+
normalizedPrompts.length > 0
|
|
302
|
+
? normalizedPrompts.map((p) => p.content).join('\n')
|
|
303
|
+
: undefined
|
|
304
|
+
|
|
305
|
+
const input = buildInteractionsInput(options.messages)
|
|
306
|
+
|
|
307
|
+
// The installed @google/genai (2.10.0) Interactions `VideoContent` type
|
|
308
|
+
// predates the `processing` field, so the structurally-built input is
|
|
309
|
+
// cast at the call boundary. The SDK forwards it to the wire unchanged.
|
|
310
|
+
const interaction = await this.client.interactions.create({
|
|
311
|
+
model,
|
|
312
|
+
...(systemInstruction !== undefined && {
|
|
313
|
+
system_instruction: systemInstruction,
|
|
314
|
+
}),
|
|
315
|
+
input: input as never,
|
|
316
|
+
})
|
|
317
|
+
|
|
318
|
+
const text = interaction.output_text ?? ''
|
|
319
|
+
|
|
320
|
+
yield {
|
|
321
|
+
type: EventType.RUN_STARTED,
|
|
322
|
+
runId,
|
|
323
|
+
threadId,
|
|
324
|
+
model,
|
|
325
|
+
timestamp: Date.now(),
|
|
326
|
+
parentRunId: options.parentRunId,
|
|
327
|
+
}
|
|
328
|
+
yield {
|
|
329
|
+
type: EventType.TEXT_MESSAGE_START,
|
|
330
|
+
messageId,
|
|
331
|
+
model,
|
|
332
|
+
timestamp: Date.now(),
|
|
333
|
+
role: 'assistant',
|
|
334
|
+
}
|
|
335
|
+
if (text) {
|
|
336
|
+
yield {
|
|
337
|
+
type: EventType.TEXT_MESSAGE_CONTENT,
|
|
338
|
+
messageId,
|
|
339
|
+
model,
|
|
340
|
+
timestamp: Date.now(),
|
|
341
|
+
delta: text,
|
|
342
|
+
content: text,
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
yield {
|
|
346
|
+
type: EventType.TEXT_MESSAGE_END,
|
|
347
|
+
messageId,
|
|
348
|
+
model,
|
|
349
|
+
timestamp: Date.now(),
|
|
350
|
+
}
|
|
351
|
+
yield {
|
|
352
|
+
type: EventType.RUN_FINISHED,
|
|
353
|
+
runId,
|
|
354
|
+
threadId,
|
|
355
|
+
model,
|
|
356
|
+
timestamp: Date.now(),
|
|
357
|
+
finishReason: 'stop',
|
|
358
|
+
}
|
|
359
|
+
} catch (error) {
|
|
360
|
+
const rawEvent = toRunErrorRawEvent(error)
|
|
361
|
+
logger.errors('gemini.interactionsStream fatal', {
|
|
362
|
+
error,
|
|
363
|
+
source: 'gemini.interactionsStream',
|
|
364
|
+
})
|
|
365
|
+
yield {
|
|
366
|
+
type: EventType.RUN_ERROR,
|
|
367
|
+
model,
|
|
368
|
+
timestamp: Date.now(),
|
|
369
|
+
message:
|
|
370
|
+
error instanceof Error
|
|
371
|
+
? error.message
|
|
372
|
+
: 'An unknown error occurred during the chat stream.',
|
|
373
|
+
...(rawEvent !== undefined && { rawEvent }),
|
|
374
|
+
error: {
|
|
375
|
+
message:
|
|
376
|
+
error instanceof Error
|
|
377
|
+
? error.message
|
|
378
|
+
: 'An unknown error occurred during the chat stream.',
|
|
379
|
+
},
|
|
380
|
+
}
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
|
|
164
384
|
/**
|
|
165
385
|
* Generate structured output using Gemini's native JSON response format.
|
|
166
386
|
* Uses responseMimeType: 'application/json' and responseSchema for structured output.
|
|
@@ -595,29 +815,42 @@ export class GeminiTextAdapter<
|
|
|
595
815
|
case 'audio':
|
|
596
816
|
case 'video':
|
|
597
817
|
case 'document': {
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
818
|
+
const geminiPart: Part =
|
|
819
|
+
part.source.type === 'data'
|
|
820
|
+
? {
|
|
821
|
+
inlineData: {
|
|
822
|
+
data: part.source.value,
|
|
823
|
+
mimeType: part.source.mimeType,
|
|
824
|
+
},
|
|
825
|
+
}
|
|
826
|
+
: {
|
|
827
|
+
fileData: {
|
|
828
|
+
fileUri: part.source.value,
|
|
829
|
+
// For URL sources, use provided mimeType or fall back to
|
|
830
|
+
// reasonable defaults.
|
|
831
|
+
mimeType:
|
|
832
|
+
part.source.mimeType ?? DEFAULT_MEDIA_MIME_TYPES[part.type],
|
|
833
|
+
},
|
|
834
|
+
}
|
|
835
|
+
|
|
836
|
+
// Apply single-pass video sampling controls (fps / clip offsets) from
|
|
837
|
+
// the part metadata. `processing: 'agentic'` is handled separately via
|
|
838
|
+
// the Interactions API and never reaches this generateContent path.
|
|
839
|
+
if (part.type === 'video') {
|
|
840
|
+
const meta = part.metadata as GeminiVideoMetadata | undefined
|
|
841
|
+
const videoMetadata: VideoMetadata = {
|
|
842
|
+
...(meta?.fps !== undefined && { fps: meta.fps }),
|
|
843
|
+
...(meta?.startOffset !== undefined && {
|
|
844
|
+
startOffset: meta.startOffset,
|
|
845
|
+
}),
|
|
846
|
+
...(meta?.endOffset !== undefined && { endOffset: meta.endOffset }),
|
|
604
847
|
}
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
const defaultMimeType = {
|
|
608
|
-
image: 'image/jpeg',
|
|
609
|
-
audio: 'audio/mp3',
|
|
610
|
-
video: 'video/mp4',
|
|
611
|
-
document: 'application/pdf',
|
|
612
|
-
}[part.type]
|
|
613
|
-
|
|
614
|
-
return {
|
|
615
|
-
fileData: {
|
|
616
|
-
fileUri: part.source.value,
|
|
617
|
-
mimeType: part.source.mimeType ?? defaultMimeType,
|
|
618
|
-
},
|
|
848
|
+
if (Object.keys(videoMetadata).length > 0) {
|
|
849
|
+
geminiPart.videoMetadata = videoMetadata
|
|
619
850
|
}
|
|
620
851
|
}
|
|
852
|
+
|
|
853
|
+
return geminiPart
|
|
621
854
|
}
|
|
622
855
|
default: {
|
|
623
856
|
const _exhaustiveCheck: never = part
|
|
@@ -13,6 +13,7 @@ import {
|
|
|
13
13
|
import { assertUniqueToolNames } from '@tanstack/ai/adapter-internals'
|
|
14
14
|
import type { InternalLogger } from '@tanstack/ai/adapter-internals'
|
|
15
15
|
import type {
|
|
16
|
+
GeminiChatModelProviderOptionsByName,
|
|
16
17
|
GeminiChatModelToolCapabilitiesByName,
|
|
17
18
|
GeminiModelInputModalitiesByName,
|
|
18
19
|
GeminiModels,
|
|
@@ -104,15 +105,31 @@ type ToolCallState = {
|
|
|
104
105
|
// ===========================
|
|
105
106
|
|
|
106
107
|
/**
|
|
107
|
-
* Resolve provider options for a specific model.
|
|
108
|
-
*
|
|
109
|
-
*
|
|
110
|
-
*
|
|
111
|
-
*
|
|
112
|
-
*
|
|
113
|
-
* signature.
|
|
108
|
+
* Resolve provider options for a specific model. Reuses the chat-model
|
|
109
|
+
* options map from `model-meta.ts` so a model's allowed thinking levels are
|
|
110
|
+
* declared once: the Interactions API takes the same levels as
|
|
111
|
+
* `generateContent` but lowercased (`low`, not `LOW`). Models whose chat
|
|
112
|
+
* options carry no `thinkingConfig` resolve to `never`, so `thinking_level`
|
|
113
|
+
* can't be set on them. Everything else falls through to the flat SDK shape.
|
|
114
114
|
*/
|
|
115
|
-
type
|
|
115
|
+
type InteractionsThinkingLevel<TModel> =
|
|
116
|
+
TModel extends keyof GeminiChatModelProviderOptionsByName
|
|
117
|
+
? GeminiChatModelProviderOptionsByName[TModel] extends {
|
|
118
|
+
thinkingConfig?: { thinkingLevel?: infer L extends string }
|
|
119
|
+
}
|
|
120
|
+
? Lowercase<L>
|
|
121
|
+
: never
|
|
122
|
+
: never
|
|
123
|
+
|
|
124
|
+
type ResolveProviderOptions<TModel extends GeminiModels> = Omit<
|
|
125
|
+
GeminiTextInteractionsProviderOptions,
|
|
126
|
+
'generation_config'
|
|
127
|
+
> & {
|
|
128
|
+
generation_config?: Omit<
|
|
129
|
+
NonNullable<GeminiTextInteractionsProviderOptions['generation_config']>,
|
|
130
|
+
'thinking_level'
|
|
131
|
+
> & { thinking_level?: InteractionsThinkingLevel<TModel> }
|
|
132
|
+
}
|
|
116
133
|
|
|
117
134
|
/**
|
|
118
135
|
* Resolve input modalities for a specific model. Reuses the chat-model
|
|
@@ -169,7 +186,7 @@ type ResolveToolCapabilities<TModel extends string> =
|
|
|
169
186
|
*/
|
|
170
187
|
export class GeminiTextInteractionsAdapter<
|
|
171
188
|
TModel extends GeminiModels,
|
|
172
|
-
TProviderOptions extends Record<string, any> = ResolveProviderOptions
|
|
189
|
+
TProviderOptions extends Record<string, any> = ResolveProviderOptions<TModel>,
|
|
173
190
|
TInputModalities extends ReadonlyArray<Modality> =
|
|
174
191
|
ResolveInputModalities<TModel>,
|
|
175
192
|
TToolCapabilities extends ReadonlyArray<string> =
|
|
@@ -455,7 +472,7 @@ export function createGeminiTextInteractions<TModel extends GeminiModels>(
|
|
|
455
472
|
config?: Omit<GeminiTextInteractionsConfig, 'apiKey'>,
|
|
456
473
|
): GeminiTextInteractionsAdapter<
|
|
457
474
|
TModel,
|
|
458
|
-
ResolveProviderOptions
|
|
475
|
+
ResolveProviderOptions<TModel>,
|
|
459
476
|
ResolveInputModalities<TModel>,
|
|
460
477
|
ResolveToolCapabilities<TModel>
|
|
461
478
|
> {
|
|
@@ -468,7 +485,7 @@ export function geminiTextInteractions<TModel extends GeminiModels>(
|
|
|
468
485
|
config?: Omit<GeminiTextInteractionsConfig, 'apiKey'>,
|
|
469
486
|
): GeminiTextInteractionsAdapter<
|
|
470
487
|
TModel,
|
|
471
|
-
ResolveProviderOptions
|
|
488
|
+
ResolveProviderOptions<TModel>,
|
|
472
489
|
ResolveInputModalities<TModel>,
|
|
473
490
|
ResolveToolCapabilities<TModel>
|
|
474
491
|
> {
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
import { FileState } from '@google/genai'
|
|
2
|
+
import { createGeminiClient, getGeminiApiKeyFromEnv } from '../utils'
|
|
3
|
+
import type { GeminiVideoMetadata } from '../message-types'
|
|
4
|
+
import type { VideoPart } from '@tanstack/ai'
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* A file uploaded to the Gemini Files API and ready to reference in a message.
|
|
8
|
+
*/
|
|
9
|
+
export interface GeminiUploadedFile {
|
|
10
|
+
/** Resource name, e.g. `"files/abc123"`. */
|
|
11
|
+
name: string
|
|
12
|
+
/** File URI to reference from message content (as a `url` source). */
|
|
13
|
+
uri: string
|
|
14
|
+
/** MIME type reported by the Files API. */
|
|
15
|
+
mimeType: string
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* Options for {@link uploadGeminiFile}.
|
|
20
|
+
*/
|
|
21
|
+
export interface GeminiUploadFileOptions {
|
|
22
|
+
/**
|
|
23
|
+
* API key. Falls back to `GOOGLE_API_KEY` / `GEMINI_API_KEY` from the
|
|
24
|
+
* environment when omitted.
|
|
25
|
+
*/
|
|
26
|
+
apiKey?: string
|
|
27
|
+
/**
|
|
28
|
+
* MIME type of the file (e.g. `"video/mp4"`). Recommended so the Files API
|
|
29
|
+
* processes and serves the file with the correct type.
|
|
30
|
+
*/
|
|
31
|
+
mimeType?: string
|
|
32
|
+
/** Poll interval while the file is `PROCESSING`, in ms. Default `5000`. */
|
|
33
|
+
pollIntervalMs?: number
|
|
34
|
+
/** Max time to wait for processing, in ms. Default `300000` (5 min). */
|
|
35
|
+
timeoutMs?: number
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Upload a file via the Gemini Files API and wait until it is `ACTIVE`.
|
|
40
|
+
*
|
|
41
|
+
* Large media (notably video) must be uploaded rather than inlined as base64;
|
|
42
|
+
* the Files API processes uploads asynchronously. This wraps the upload +
|
|
43
|
+
* poll-until-ready loop and returns a reference you can drop into message
|
|
44
|
+
* content as a `url` source (see {@link geminiVideoPart}).
|
|
45
|
+
*
|
|
46
|
+
* @throws if the upload has no URI, or processing fails or times out.
|
|
47
|
+
*/
|
|
48
|
+
export async function uploadGeminiFile(
|
|
49
|
+
file: string | Blob,
|
|
50
|
+
options: GeminiUploadFileOptions = {},
|
|
51
|
+
): Promise<GeminiUploadedFile> {
|
|
52
|
+
const {
|
|
53
|
+
apiKey = getGeminiApiKeyFromEnv(),
|
|
54
|
+
mimeType,
|
|
55
|
+
pollIntervalMs = 5000,
|
|
56
|
+
timeoutMs = 300_000,
|
|
57
|
+
} = options
|
|
58
|
+
|
|
59
|
+
const client = createGeminiClient({ apiKey })
|
|
60
|
+
|
|
61
|
+
let uploaded = await client.files.upload({
|
|
62
|
+
file,
|
|
63
|
+
...(mimeType && { config: { mimeType } }),
|
|
64
|
+
})
|
|
65
|
+
|
|
66
|
+
const fileName = uploaded.name
|
|
67
|
+
if (!fileName) {
|
|
68
|
+
throw new Error('Gemini file upload did not return a file name.')
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
const deadline = Date.now() + timeoutMs
|
|
72
|
+
while (uploaded.state === FileState.PROCESSING) {
|
|
73
|
+
if (Date.now() > deadline) {
|
|
74
|
+
throw new Error(
|
|
75
|
+
`Gemini file processing timed out after ${timeoutMs}ms (${fileName}).`,
|
|
76
|
+
)
|
|
77
|
+
}
|
|
78
|
+
await new Promise((resolve) => setTimeout(resolve, pollIntervalMs))
|
|
79
|
+
uploaded = await client.files.get({ name: fileName })
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
if (uploaded.state === FileState.FAILED) {
|
|
83
|
+
throw new Error(
|
|
84
|
+
`Gemini file processing failed: ${uploaded.error?.message ?? String(uploaded.state)}`,
|
|
85
|
+
)
|
|
86
|
+
}
|
|
87
|
+
if (!uploaded.uri) {
|
|
88
|
+
throw new Error('Gemini file upload did not return a URI.')
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
return {
|
|
92
|
+
name: fileName,
|
|
93
|
+
uri: uploaded.uri,
|
|
94
|
+
mimeType: uploaded.mimeType ?? mimeType ?? 'application/octet-stream',
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Build a TanStack AI video content part from an uploaded Gemini file.
|
|
100
|
+
*
|
|
101
|
+
* Pass `metadata` to control understanding — e.g.
|
|
102
|
+
* `{ processing: 'agentic' }` to route through the agentic Interactions path,
|
|
103
|
+
* or `{ fps, startOffset, endOffset }` for single-pass sampling controls.
|
|
104
|
+
*/
|
|
105
|
+
export function geminiVideoPart(
|
|
106
|
+
file: GeminiUploadedFile,
|
|
107
|
+
metadata?: GeminiVideoMetadata,
|
|
108
|
+
): VideoPart<GeminiVideoMetadata> {
|
|
109
|
+
return {
|
|
110
|
+
type: 'video',
|
|
111
|
+
source: { type: 'url', value: file.uri, mimeType: file.mimeType },
|
|
112
|
+
...(metadata && { metadata }),
|
|
113
|
+
}
|
|
114
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -60,6 +60,14 @@ export type {
|
|
|
60
60
|
// having to add `@google/genai` to their own dependencies.
|
|
61
61
|
export { HarmBlockThreshold, HarmCategory } from '@google/genai'
|
|
62
62
|
|
|
63
|
+
// Files API helpers — upload + poll a file (e.g. video) until it is ACTIVE
|
|
64
|
+
export {
|
|
65
|
+
uploadGeminiFile,
|
|
66
|
+
geminiVideoPart,
|
|
67
|
+
type GeminiUploadedFile,
|
|
68
|
+
type GeminiUploadFileOptions,
|
|
69
|
+
} from './files/index'
|
|
70
|
+
|
|
63
71
|
// Embedding adapter - for embedding vectors
|
|
64
72
|
export {
|
|
65
73
|
GeminiEmbeddingAdapter,
|
|
@@ -168,6 +176,7 @@ export type {
|
|
|
168
176
|
GeminiImageMetadata,
|
|
169
177
|
GeminiAudioMetadata,
|
|
170
178
|
GeminiVideoMetadata,
|
|
179
|
+
GeminiVideoProcessing,
|
|
171
180
|
GeminiDocumentMetadata,
|
|
172
181
|
GeminiMessageMetadataByModality,
|
|
173
182
|
} from './message-types'
|
package/src/message-types.ts
CHANGED
|
@@ -87,6 +87,19 @@ export interface GeminiAudioMetadata {
|
|
|
87
87
|
mimeType?: GeminiAudioMimeType
|
|
88
88
|
}
|
|
89
89
|
|
|
90
|
+
/**
|
|
91
|
+
* How Gemini processes a video for understanding.
|
|
92
|
+
*
|
|
93
|
+
* - `static` (default): single-pass frame sampling via `generateContent`.
|
|
94
|
+
* - `agentic`: multi-pass "agentic" video understanding, GA on the
|
|
95
|
+
* `agentic_video`-capable flash models (`gemini-3.8-flash`, `gemini-3.7-flash`,
|
|
96
|
+
* `gemini-3.6-flash`, `gemini-3.5-flash-lite`). The text adapter routes the
|
|
97
|
+
* request through the Interactions API instead of `generateContent`. With
|
|
98
|
+
* `agentic`, the sampling rate is expressed in the text prompt (e.g. "watch
|
|
99
|
+
* it at 0.5 fps"), not via `fps`.
|
|
100
|
+
*/
|
|
101
|
+
export type GeminiVideoProcessing = 'agentic' | 'static'
|
|
102
|
+
|
|
90
103
|
/**
|
|
91
104
|
* Metadata for Gemini video content parts.
|
|
92
105
|
*/
|
|
@@ -98,6 +111,30 @@ export interface GeminiVideoMetadata {
|
|
|
98
111
|
* @see https://ai.google.dev/gemini-api/docs/vision#video-requirements
|
|
99
112
|
*/
|
|
100
113
|
mimeType?: GeminiVideoMimeType
|
|
114
|
+
/**
|
|
115
|
+
* How the model processes this video for understanding. When set, the
|
|
116
|
+
* adapter routes the request through the Gemini Interactions API. Omit for
|
|
117
|
+
* the default single-pass `generateContent` behavior.
|
|
118
|
+
*/
|
|
119
|
+
processing?: GeminiVideoProcessing
|
|
120
|
+
/**
|
|
121
|
+
* Frame-rate sampling density (frames per second) for single-pass
|
|
122
|
+
* (`generateContent`) understanding. Valid range (0, 24]; defaults to 1.0
|
|
123
|
+
* on the server. Ignored when `processing` is `agentic`.
|
|
124
|
+
*
|
|
125
|
+
* @see https://ai.google.dev/gemini-api/docs/vision#customize-frame-rate
|
|
126
|
+
*/
|
|
127
|
+
fps?: number
|
|
128
|
+
/**
|
|
129
|
+
* Clip start offset, as a decimal number of seconds with an `s` suffix
|
|
130
|
+
* (e.g. `"10.5s"`). Restricts understanding to a segment of the video.
|
|
131
|
+
*/
|
|
132
|
+
startOffset?: string
|
|
133
|
+
/**
|
|
134
|
+
* Clip end offset, as a decimal number of seconds with an `s` suffix
|
|
135
|
+
* (e.g. `"45s"`). Restricts understanding to a segment of the video.
|
|
136
|
+
*/
|
|
137
|
+
endOffset?: string
|
|
101
138
|
}
|
|
102
139
|
|
|
103
140
|
/**
|
package/src/model-meta.ts
CHANGED
|
@@ -14,6 +14,7 @@ interface ModelMeta<TProviderOptions = unknown> {
|
|
|
14
14
|
input: Array<'text' | 'image' | 'audio' | 'video' | 'document'>
|
|
15
15
|
output: Array<'text' | 'image' | 'audio' | 'video'>
|
|
16
16
|
capabilities?: Array<
|
|
17
|
+
| 'agentic_video'
|
|
17
18
|
| 'audio_generation'
|
|
18
19
|
| 'batch_api'
|
|
19
20
|
| 'caching'
|
|
@@ -870,6 +871,49 @@ const GEMINI_OMNI_FLASH_PREVIEW = {
|
|
|
870
871
|
GeminiCachedContentOptions
|
|
871
872
|
>
|
|
872
873
|
|
|
874
|
+
const GEMINI_3_8_FLASH = {
|
|
875
|
+
name: 'gemini-3.8-flash',
|
|
876
|
+
max_input_tokens: 1_048_576,
|
|
877
|
+
max_output_tokens: 65_536,
|
|
878
|
+
knowledge_cutoff: '2026-03-01',
|
|
879
|
+
supports: {
|
|
880
|
+
input: ['text', 'image', 'video', 'audio', 'document'],
|
|
881
|
+
output: ['text'],
|
|
882
|
+
capabilities: [
|
|
883
|
+
'agentic_video',
|
|
884
|
+
'batch_api',
|
|
885
|
+
'caching',
|
|
886
|
+
'function_calling',
|
|
887
|
+
'structured_output',
|
|
888
|
+
'thinking',
|
|
889
|
+
],
|
|
890
|
+
tools: [
|
|
891
|
+
'code_execution',
|
|
892
|
+
'file_search',
|
|
893
|
+
'google_search',
|
|
894
|
+
'google_maps',
|
|
895
|
+
'url_context',
|
|
896
|
+
'computer_use',
|
|
897
|
+
],
|
|
898
|
+
},
|
|
899
|
+
pricing: {
|
|
900
|
+
input: {
|
|
901
|
+
normal: 0.75,
|
|
902
|
+
cached: 0.075,
|
|
903
|
+
},
|
|
904
|
+
output: {
|
|
905
|
+
normal: 3.75,
|
|
906
|
+
},
|
|
907
|
+
},
|
|
908
|
+
} as const satisfies ModelMeta<
|
|
909
|
+
GeminiToolConfigOptions &
|
|
910
|
+
GeminiSafetyOptions &
|
|
911
|
+
GeminiCommonConfigOptions &
|
|
912
|
+
GeminiCachedContentOptions &
|
|
913
|
+
GeminiStructuredOutputOptions &
|
|
914
|
+
GeminiThinkingOptions<'LOW' | 'MEDIUM' | 'HIGH'>
|
|
915
|
+
>
|
|
916
|
+
|
|
873
917
|
const GEMINI_3_7_FLASH = {
|
|
874
918
|
name: 'gemini-3.7-flash',
|
|
875
919
|
max_input_tokens: 1_048_576,
|
|
@@ -879,6 +923,7 @@ const GEMINI_3_7_FLASH = {
|
|
|
879
923
|
input: ['text', 'image', 'video', 'audio', 'document'],
|
|
880
924
|
output: ['text'],
|
|
881
925
|
capabilities: [
|
|
926
|
+
'agentic_video',
|
|
882
927
|
'batch_api',
|
|
883
928
|
'caching',
|
|
884
929
|
'function_calling',
|
|
@@ -923,6 +968,7 @@ const GEMINI_3_6_FLASH = {
|
|
|
923
968
|
input: ['text', 'image', 'video', 'audio', 'document'],
|
|
924
969
|
output: ['text'],
|
|
925
970
|
capabilities: [
|
|
971
|
+
'agentic_video',
|
|
926
972
|
'batch_api',
|
|
927
973
|
'caching',
|
|
928
974
|
'function_calling',
|
|
@@ -1006,6 +1052,7 @@ const GEMINI_3_5_FLASH_LITE = {
|
|
|
1006
1052
|
input: ['text', 'image', 'video', 'audio', 'document'],
|
|
1007
1053
|
output: ['text'],
|
|
1008
1054
|
capabilities: [
|
|
1055
|
+
'agentic_video',
|
|
1009
1056
|
'batch_api',
|
|
1010
1057
|
'caching',
|
|
1011
1058
|
'function_calling',
|
|
@@ -1039,6 +1086,7 @@ const GEMINI_3_5_FLASH_LITE = {
|
|
|
1039
1086
|
>
|
|
1040
1087
|
|
|
1041
1088
|
export const GEMINI_MODELS = [
|
|
1089
|
+
GEMINI_3_8_FLASH.name,
|
|
1042
1090
|
GEMINI_3_7_FLASH.name,
|
|
1043
1091
|
GEMINI_3_6_FLASH.name,
|
|
1044
1092
|
GEMINI_3_5_FLASH.name,
|
|
@@ -1060,6 +1108,7 @@ export const GEMINI_MODELS = [
|
|
|
1060
1108
|
* brittle and keeps the engine's legacy finalization fallback.
|
|
1061
1109
|
*/
|
|
1062
1110
|
export const GEMINI_COMBINED_TOOLS_AND_SCHEMA_MODELS = new Set<string>([
|
|
1111
|
+
GEMINI_3_8_FLASH.name,
|
|
1063
1112
|
GEMINI_3_7_FLASH.name,
|
|
1064
1113
|
GEMINI_3_6_FLASH.name,
|
|
1065
1114
|
GEMINI_3_5_FLASH.name,
|
|
@@ -1217,6 +1266,12 @@ export type GeminiEmbeddingModelInputModalitiesByName = {
|
|
|
1217
1266
|
// Manual type map for per-model provider options
|
|
1218
1267
|
export type GeminiChatModelProviderOptionsByName = {
|
|
1219
1268
|
// Models with thinking and structured output support
|
|
1269
|
+
[GEMINI_3_8_FLASH.name]: GeminiToolConfigOptions &
|
|
1270
|
+
GeminiSafetyOptions &
|
|
1271
|
+
GeminiCommonConfigOptions &
|
|
1272
|
+
GeminiCachedContentOptions &
|
|
1273
|
+
GeminiStructuredOutputOptions &
|
|
1274
|
+
GeminiThinkingOptions<'LOW' | 'MEDIUM' | 'HIGH'>
|
|
1220
1275
|
[GEMINI_3_7_FLASH.name]: GeminiToolConfigOptions &
|
|
1221
1276
|
GeminiSafetyOptions &
|
|
1222
1277
|
GeminiCommonConfigOptions &
|
|
@@ -1290,6 +1345,7 @@ export type GeminiChatModelProviderOptionsByName = {
|
|
|
1290
1345
|
* Based on the 'supports.tools' arrays defined for each model.
|
|
1291
1346
|
*/
|
|
1292
1347
|
export type GeminiChatModelToolCapabilitiesByName = {
|
|
1348
|
+
[GEMINI_3_8_FLASH.name]: typeof GEMINI_3_8_FLASH.supports.tools
|
|
1293
1349
|
[GEMINI_3_7_FLASH.name]: typeof GEMINI_3_7_FLASH.supports.tools
|
|
1294
1350
|
[GEMINI_3_6_FLASH.name]: typeof GEMINI_3_6_FLASH.supports.tools
|
|
1295
1351
|
[GEMINI_3_5_FLASH.name]: typeof GEMINI_3_5_FLASH.supports.tools
|
|
@@ -1318,6 +1374,7 @@ export type GeminiChatModelToolCapabilitiesByName = {
|
|
|
1318
1374
|
*/
|
|
1319
1375
|
export type GeminiModelInputModalitiesByName = {
|
|
1320
1376
|
// Models with full multimodal support (text, image, audio, video, document)
|
|
1377
|
+
[GEMINI_3_8_FLASH.name]: typeof GEMINI_3_8_FLASH.supports.input
|
|
1321
1378
|
[GEMINI_3_7_FLASH.name]: typeof GEMINI_3_7_FLASH.supports.input
|
|
1322
1379
|
[GEMINI_3_6_FLASH.name]: typeof GEMINI_3_6_FLASH.supports.input
|
|
1323
1380
|
[GEMINI_3_5_FLASH.name]: typeof GEMINI_3_5_FLASH.supports.input
|