@tanstack/ai-gemini 0.19.1 → 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/dist/esm/adapters/audio.d.ts +1 -1
  2. package/dist/esm/adapters/audio.js.map +1 -1
  3. package/dist/esm/adapters/image.d.ts +1 -1
  4. package/dist/esm/adapters/image.js +17 -39
  5. package/dist/esm/adapters/image.js.map +1 -1
  6. package/dist/esm/adapters/summarize.d.ts +1 -1
  7. package/dist/esm/adapters/summarize.js.map +1 -1
  8. package/dist/esm/adapters/text.d.ts +1 -1
  9. package/dist/esm/adapters/text.js.map +1 -1
  10. package/dist/esm/adapters/tts.d.ts +1 -1
  11. package/dist/esm/adapters/tts.js.map +1 -1
  12. package/dist/esm/adapters/video.d.ts +60 -11
  13. package/dist/esm/adapters/video.js +205 -6
  14. package/dist/esm/adapters/video.js.map +1 -1
  15. package/dist/esm/experimental/text-interactions/adapter.d.ts +1 -1
  16. package/dist/esm/experimental/text-interactions/adapter.js.map +1 -1
  17. package/dist/esm/index.d.ts +6 -3
  18. package/dist/esm/index.js +9 -3
  19. package/dist/esm/index.js.map +1 -1
  20. package/dist/esm/model-meta.d.ts +11 -3
  21. package/dist/esm/model-meta.js +9 -1
  22. package/dist/esm/model-meta.js.map +1 -1
  23. package/dist/esm/realtime/adapter.d.ts +22 -0
  24. package/dist/esm/realtime/adapter.js +233 -0
  25. package/dist/esm/realtime/adapter.js.map +1 -0
  26. package/dist/esm/realtime/client.d.ts +98 -0
  27. package/dist/esm/realtime/client.js +389 -0
  28. package/dist/esm/realtime/client.js.map +1 -0
  29. package/dist/esm/realtime/index.d.ts +3 -0
  30. package/dist/esm/realtime/token.d.ts +26 -0
  31. package/dist/esm/realtime/token.js +39 -0
  32. package/dist/esm/realtime/token.js.map +1 -0
  33. package/dist/esm/realtime/types.d.ts +51 -0
  34. package/dist/esm/realtime/utils.d.ts +40 -0
  35. package/dist/esm/realtime/utils.js +350 -0
  36. package/dist/esm/realtime/utils.js.map +1 -0
  37. package/dist/esm/video/video-provider-options.d.ts +59 -14
  38. package/dist/esm/video/video-provider-options.js +15 -2
  39. package/dist/esm/video/video-provider-options.js.map +1 -1
  40. package/package.json +4 -4
  41. package/src/adapters/audio.ts +1 -1
  42. package/src/adapters/image.ts +25 -49
  43. package/src/adapters/summarize.ts +1 -1
  44. package/src/adapters/text.ts +1 -1
  45. package/src/adapters/tts.ts +1 -1
  46. package/src/adapters/video.ts +333 -16
  47. package/src/experimental/text-interactions/adapter.ts +2 -2
  48. package/src/index.ts +20 -2
  49. package/src/model-meta.ts +45 -2
  50. package/src/realtime/adapter.ts +311 -0
  51. package/src/realtime/client.ts +547 -0
  52. package/src/realtime/index.ts +14 -0
  53. package/src/realtime/token.ts +70 -0
  54. package/src/realtime/types.ts +94 -0
  55. package/src/realtime/utils.ts +439 -0
  56. package/src/video/video-provider-options.ts +95 -15
@@ -0,0 +1,94 @@
1
+ import type {
2
+ ContextWindowCompressionConfig,
3
+ LiveConnectConstraints,
4
+ ThinkingConfig,
5
+ } from '@google/genai'
6
+
7
+ /**
8
+ * Gemini realtime voice options
9
+ */
10
+ export type GeminiRealtimeVoice =
11
+ | 'Achernar'
12
+ | 'Achird'
13
+ | 'Algenib'
14
+ | 'Algieba'
15
+ | 'Alnilam'
16
+ | 'Aoede'
17
+ | 'Autonoe'
18
+ | 'Callirrhoe'
19
+ | 'Charon'
20
+ | 'Despina'
21
+ | 'Enceladus'
22
+ | 'Erinome'
23
+ | 'Fenrir'
24
+ | 'Gacrux'
25
+ | 'Iapetus'
26
+ | 'Kore'
27
+ | 'Laomedeia'
28
+ | 'Leda'
29
+ | 'Orus'
30
+ | 'Pulcherrima'
31
+ | 'Puck'
32
+ | 'Rasalgethi'
33
+ | 'Sadachbia'
34
+ | 'Sadaltager'
35
+ | 'Schedar'
36
+ | 'Sulafat'
37
+ | 'Umbriel'
38
+ | 'Vindemiatrix'
39
+ | 'Zephyr'
40
+ | 'Zubenelgenubi'
41
+
42
+ /**
43
+ * Gemini realtime model options
44
+ */
45
+ export type GeminiRealtimeModel = 'gemini-3.1-flash-live-preview'
46
+
47
+ /**
48
+ * Options for the Gemini realtime client adapter
49
+ */
50
+ export interface GeminiRealtimeOptions {
51
+ /** Connection mode (default: 'websocket' in browser) */
52
+ connectionMode?: 'websocket'
53
+ model?: GeminiRealtimeModel
54
+ }
55
+
56
+ export interface StrictLiveConnectionConstraints extends Omit<
57
+ LiveConnectConstraints,
58
+ 'model'
59
+ > {
60
+ model?: GeminiRealtimeModel
61
+ }
62
+
63
+ /**
64
+ * Options for the Gemini realtime token adapter
65
+ */
66
+ export interface GeminiRealtimeTokenOptions {
67
+ expiresAt?: number
68
+ uses?: number
69
+ /**
70
+ * Config for LiveConnectConstraints for Auth Token creation.
71
+ *
72
+ * NOTE: Adding liveConnectConstraints will cause the API to ignore any config passed later to WebSocket.
73
+ */
74
+ liveConnectConstraints?: StrictLiveConnectionConstraints
75
+ }
76
+
77
+ /**
78
+ * Gemini-specific realtime options, passed through `providerOptions` on a
79
+ * realtime session config.
80
+ */
81
+ export interface GeminiRealtimeProviderOptions {
82
+ /** Enable Google Search grounding (mutually exclusive with custom tools). */
83
+ googleGrounding?: boolean
84
+ /** Enable proactive audio so the model may choose when to respond. */
85
+ proactiveAudio?: boolean
86
+ /** Enable affective (emotion-aware) dialog. */
87
+ enableAffectiveDialog?: boolean
88
+ /** Context window compression configuration. */
89
+ contextWindowCompression?: ContextWindowCompressionConfig
90
+ /** Thinking configuration. */
91
+ thinkingConfig?: ThinkingConfig
92
+ /** BCP-47 language code for speech output. */
93
+ languageCode?: string
94
+ }
@@ -0,0 +1,439 @@
1
+ import type { GeminiLiveClient } from './client'
2
+
3
+ /**
4
+ * Audio Worklet Processor for capturing and processing audio
5
+ */
6
+ const captureWorkletCode = `
7
+ class AudioCaptureProcessor extends AudioWorkletProcessor {
8
+ constructor() {
9
+ super();
10
+ this.bufferSize = 512; // 32ms at 16kHz — per Gemini best practices (20-40ms chunks)
11
+ this.buffer = new Float32Array(this.bufferSize);
12
+ this.bufferIndex = 0;
13
+ }
14
+
15
+ process(inputs, outputs, parameters) {
16
+ const input = inputs[0];
17
+
18
+ if (input && input.length > 0) {
19
+ const inputChannel = input[0];
20
+
21
+ // Buffer the incoming audio
22
+ for (let i = 0; i < inputChannel.length; i++) {
23
+ this.buffer[this.bufferIndex++] = inputChannel[i];
24
+
25
+ // When buffer is full, send it to main thread
26
+ if (this.bufferIndex >= this.bufferSize) {
27
+ // Send the buffered audio to the main thread
28
+ this.port.postMessage({
29
+ type: "audio",
30
+ data: this.buffer.slice(),
31
+ });
32
+
33
+ // Reset buffer
34
+ this.bufferIndex = 0;
35
+ }
36
+ }
37
+ }
38
+
39
+ // Return true to keep the processor alive
40
+ return true;
41
+ }
42
+ }
43
+
44
+ // Register the processor
45
+ registerProcessor("audio-capture-processor", AudioCaptureProcessor);`
46
+
47
+ /**
48
+ * Audio Playback Worklet Processor for playing PCM audio.
49
+ * Uses an offset tracker instead of slice() to avoid allocations
50
+ * on the real-time audio thread.
51
+ */
52
+ const playbackWorkletCode = `
53
+ class PCMProcessor extends AudioWorkletProcessor {
54
+ constructor() {
55
+ super();
56
+ this.audioQueue = [];
57
+ this.currentOffset = 0; // Track position in current buffer (avoids slice())
58
+
59
+ this.port.onmessage = (event) => {
60
+ if (event.data === "interrupt") {
61
+ // Clear the queue on interrupt
62
+ this.audioQueue = [];
63
+ this.currentOffset = 0;
64
+ } else if (event.data instanceof Float32Array) {
65
+ // Add audio data to the queue
66
+ this.audioQueue.push(event.data);
67
+ }
68
+ };
69
+ }
70
+
71
+ process(inputs, outputs, parameters) {
72
+ const output = outputs[0];
73
+ if (output.length === 0) return true;
74
+
75
+ const channel = output[0];
76
+ let outputIndex = 0;
77
+
78
+ // Fill the output buffer from the queue
79
+ while (outputIndex < channel.length && this.audioQueue.length > 0) {
80
+ const currentBuffer = this.audioQueue[0];
81
+
82
+ if (!currentBuffer || currentBuffer.length === 0) {
83
+ this.audioQueue.shift();
84
+ this.currentOffset = 0;
85
+ continue;
86
+ }
87
+
88
+ const remainingOutput = channel.length - outputIndex;
89
+ const remainingBuffer = currentBuffer.length - this.currentOffset;
90
+ const copyLength = Math.min(remainingOutput, remainingBuffer);
91
+
92
+ // Copy audio data to output using offset (no slice allocation)
93
+ for (let i = 0; i < copyLength; i++) {
94
+ channel[outputIndex++] = currentBuffer[this.currentOffset++];
95
+ }
96
+
97
+ // If we've consumed the entire buffer, move to the next one
98
+ if (this.currentOffset >= currentBuffer.length) {
99
+ this.audioQueue.shift();
100
+ this.currentOffset = 0;
101
+ }
102
+ }
103
+
104
+ // Fill remaining output with silence
105
+ while (outputIndex < channel.length) {
106
+ channel[outputIndex++] = 0;
107
+ }
108
+
109
+ return true;
110
+ }
111
+ }
112
+
113
+ registerProcessor("pcm-processor", PCMProcessor);`
114
+
115
+ function calculateLevel(analyser: AnalyserNode): number {
116
+ const data = new Uint8Array(analyser.fftSize)
117
+ analyser.getByteTimeDomainData(data)
118
+
119
+ // Find peak deviation from center (128 is silence)
120
+ // This is more responsive than RMS for voice level meters
121
+ let maxDeviation = 0
122
+ for (const sample of data) {
123
+ const deviation = Math.abs(sample - 128)
124
+ if (deviation > maxDeviation) {
125
+ maxDeviation = deviation
126
+ }
127
+ }
128
+
129
+ // Normalize to 0-1 range (max deviation is 128)
130
+ // Scale by 1.5x so that ~66% amplitude reads as full scale
131
+ // This provides good visual feedback without pegging too early
132
+ const normalized = maxDeviation / 128
133
+ return Math.min(1, normalized * 1.5)
134
+ }
135
+
136
+ export function base64ToArrayBuffer(base64: string): ArrayBuffer {
137
+ const binary = atob(base64)
138
+ const bytes = Uint8Array.from(binary, (char) => char.charCodeAt(0))
139
+ return bytes.buffer
140
+ }
141
+
142
+ // Empty arrays for when visualization isn't available
143
+ // frequencyBinCount = fftSize / 2 = 1024
144
+ const emptyFrequencyData = new Uint8Array(1024)
145
+ const emptyTimeDomainData = new Uint8Array(2048).fill(128) // 128 is silence
146
+
147
+ export class AudioStreamer {
148
+ private audioContext: AudioContext | null = null
149
+ private audioWorklet: AudioWorkletNode | null = null
150
+ private mediaStream: MediaStream | null = null
151
+ private analyser: AnalyserNode | null = null
152
+ private isStreaming = false
153
+ private readonly sampleRate = 16000
154
+ private readonly client: GeminiLiveClient | null = null
155
+
156
+ constructor(client: GeminiLiveClient) {
157
+ this.client = client
158
+ }
159
+
160
+ get inputLevel() {
161
+ if (!this.analyser) return 0
162
+ return calculateLevel(this.analyser)
163
+ }
164
+
165
+ get inputFrequencyData() {
166
+ if (!this.analyser) return emptyFrequencyData
167
+ const data = new Uint8Array(this.analyser.frequencyBinCount)
168
+ this.analyser.getByteFrequencyData(data)
169
+ return data
170
+ }
171
+
172
+ get inputTimeDomainData() {
173
+ if (!this.analyser) return emptyTimeDomainData
174
+ const data = new Uint8Array(this.analyser.fftSize)
175
+ this.analyser.getByteTimeDomainData(data)
176
+ return data
177
+ }
178
+
179
+ get inputSampleRate() {
180
+ return this.sampleRate
181
+ }
182
+
183
+ async start() {
184
+ try {
185
+ const audioConstraints: MediaTrackConstraints = {
186
+ sampleRate: this.sampleRate,
187
+ echoCancellation: true,
188
+ noiseSuppression: true,
189
+ autoGainControl: true,
190
+ }
191
+
192
+ // Get microphone access
193
+ this.mediaStream = await navigator.mediaDevices.getUserMedia({
194
+ audio: audioConstraints,
195
+ })
196
+
197
+ // Check if native AGC is active
198
+ const track = this.mediaStream.getAudioTracks()[0]
199
+ const settings = track?.getSettings()
200
+
201
+ if (settings?.autoGainControl) {
202
+ console.warn('Native AGC not supported.')
203
+ }
204
+
205
+ // Create audio context
206
+ this.audioContext = new AudioContext({
207
+ sampleRate: this.sampleRate,
208
+ })
209
+
210
+ if (this.audioContext.state === 'suspended') {
211
+ await this.audioContext.resume()
212
+ }
213
+
214
+ const workletBlob = new Blob([captureWorkletCode], {
215
+ type: 'application/javascript',
216
+ })
217
+ const workletUrl = URL.createObjectURL(workletBlob)
218
+
219
+ // Load the audio worklet module, then release the blob URL.
220
+ await this.audioContext.audioWorklet.addModule(workletUrl)
221
+ URL.revokeObjectURL(workletUrl)
222
+
223
+ // Create the audio worklet node
224
+ this.audioWorklet = new AudioWorkletNode(
225
+ this.audioContext,
226
+ 'audio-capture-processor',
227
+ )
228
+
229
+ // Set up message handling from the worklet
230
+ this.audioWorklet.port.onmessage = (event) => {
231
+ if (!this.isStreaming) return
232
+
233
+ if (event.data.type === 'audio') {
234
+ const inputData = event.data.data
235
+ const pcmData = this.convertToPCM16(inputData)
236
+ const base64Audio = this.arrayBufferToBase64(pcmData)
237
+
238
+ // Send to Gemini only if after setup complete
239
+ if (this.client?.isSetupComplete) {
240
+ this.client.sendAudioMessage(base64Audio)
241
+ }
242
+ }
243
+ }
244
+
245
+ // Create analyser for volume detection
246
+ this.analyser = this.audioContext.createAnalyser()
247
+ this.analyser.fftSize = 2048 // Larger size for more accurate level detection
248
+ this.analyser.smoothingTimeConstant = 0.3
249
+
250
+ // Connect the audio graph
251
+ const source = this.audioContext.createMediaStreamSource(this.mediaStream)
252
+ source.connect(this.analyser)
253
+ this.analyser.connect(this.audioWorklet)
254
+
255
+ // Start streaming
256
+ this.isStreaming = true
257
+ } catch (error) {
258
+ // Clean up the mic + audio context if setup failed partway through.
259
+ this.stop()
260
+ throw error
261
+ }
262
+ }
263
+
264
+ stop() {
265
+ this.isStreaming = false
266
+
267
+ if (this.audioWorklet) {
268
+ this.audioWorklet.disconnect()
269
+ this.audioWorklet.port.close()
270
+ this.audioWorklet = null
271
+ }
272
+
273
+ if (this.audioContext) {
274
+ void this.audioContext.close()
275
+ this.audioContext = null
276
+ }
277
+
278
+ if (this.mediaStream) {
279
+ this.mediaStream.getTracks().forEach((track) => track.stop())
280
+ this.mediaStream = null
281
+ }
282
+ }
283
+
284
+ startAudioCapture() {
285
+ if (this.mediaStream) {
286
+ for (const track of this.mediaStream.getAudioTracks()) {
287
+ track.enabled = true
288
+ }
289
+ }
290
+ this.isStreaming = true
291
+ }
292
+
293
+ stopAudioCapture() {
294
+ if (this.mediaStream) {
295
+ // Disable tracks rather than stopping them to allow re-enabling
296
+ for (const track of this.mediaStream.getAudioTracks()) {
297
+ track.enabled = false
298
+ }
299
+ }
300
+ this.isStreaming = false
301
+ }
302
+
303
+ private convertToPCM16(float32Array: Float32Array): ArrayBuffer {
304
+ const int16Array = new Int16Array(float32Array.length)
305
+ for (let i = 0; i < float32Array.length; i++) {
306
+ const sample = Math.max(-1, Math.min(1, float32Array[i] ?? 0))
307
+ int16Array[i] = sample * 0x7fff
308
+ }
309
+ return int16Array.buffer
310
+ }
311
+
312
+ private arrayBufferToBase64(buffer: ArrayBuffer): string {
313
+ const bytes = new Uint8Array(buffer)
314
+ const binary = String.fromCharCode(...bytes)
315
+ return btoa(binary)
316
+ }
317
+ }
318
+
319
+ export class AudioPlayer {
320
+ private audioContext: AudioContext | null = null
321
+ private workletNode: AudioWorkletNode | null = null
322
+ private gainNode: GainNode | null = null
323
+ private analyser: AnalyserNode | null = null
324
+ private isInitialized = false
325
+ private volume = 1.0
326
+ private readonly sampleRate = 24000
327
+
328
+ get outputLevel() {
329
+ if (!this.analyser) return 0
330
+ return calculateLevel(this.analyser)
331
+ }
332
+
333
+ get outputFrequencyData() {
334
+ if (!this.analyser) return emptyFrequencyData
335
+ const data = new Uint8Array(this.analyser.frequencyBinCount)
336
+ this.analyser.getByteFrequencyData(data)
337
+ return data
338
+ }
339
+
340
+ get outputTimeDomainData() {
341
+ if (!this.analyser) return emptyTimeDomainData
342
+ const data = new Uint8Array(this.analyser.fftSize)
343
+ this.analyser.getByteTimeDomainData(data)
344
+ return data
345
+ }
346
+
347
+ get outputSampleRate() {
348
+ return this.sampleRate
349
+ }
350
+
351
+ async init() {
352
+ if (this.isInitialized) return
353
+
354
+ try {
355
+ // Create audio context at 24kHz to match Gemini
356
+ this.audioContext = new AudioContext({
357
+ sampleRate: this.sampleRate,
358
+ })
359
+
360
+ const workletBlob = new Blob([playbackWorkletCode], {
361
+ type: 'application/javascript',
362
+ })
363
+ const workletUrl = URL.createObjectURL(workletBlob)
364
+
365
+ // Load the audio worklet module, then release the blob URL.
366
+ await this.audioContext.audioWorklet.addModule(workletUrl)
367
+ URL.revokeObjectURL(workletUrl)
368
+
369
+ // Create worklet node
370
+ this.workletNode = new AudioWorkletNode(
371
+ this.audioContext,
372
+ 'pcm-processor',
373
+ )
374
+
375
+ // Create gain node for volume control
376
+ this.gainNode = this.audioContext.createGain()
377
+ this.gainNode.gain.value = this.volume
378
+
379
+ // Create analyser for volume detection
380
+ this.analyser = this.audioContext.createAnalyser()
381
+ this.analyser.fftSize = 2048 // Larger size for more accurate level detection
382
+ this.analyser.smoothingTimeConstant = 0.3
383
+
384
+ // Connect nodes
385
+ this.workletNode.connect(this.gainNode)
386
+ this.gainNode.connect(this.analyser)
387
+ this.analyser.connect(this.audioContext.destination)
388
+
389
+ this.isInitialized = true
390
+ } catch (error) {
391
+ // Release the audio context if initialization failed partway through.
392
+ this.destroy()
393
+ throw error
394
+ }
395
+ }
396
+
397
+ async play(pcmData: ArrayBuffer) {
398
+ if (!this.isInitialized) {
399
+ await this.init()
400
+ }
401
+
402
+ // Resume audio context if suspended
403
+ if (this.audioContext?.state === 'suspended') {
404
+ await this.audioContext.resume()
405
+ }
406
+
407
+ // Convert PCM16 LE to Float32
408
+ const inputArray = new Int16Array(pcmData)
409
+ const float32Data = new Float32Array(inputArray.length)
410
+ for (let i = 0; i < inputArray.length; i++) {
411
+ float32Data[i] = (inputArray[i] ?? 0) / 32768
412
+ }
413
+
414
+ // Send to worklet for playback
415
+ this.workletNode?.port.postMessage(float32Data)
416
+ }
417
+
418
+ /* Interrupt playback */
419
+ interrupt() {
420
+ if (this.workletNode) {
421
+ this.workletNode.port.postMessage('interrupt')
422
+ }
423
+ }
424
+
425
+ setVolume(volume: number) {
426
+ this.volume = Math.max(0, Math.min(1, volume))
427
+ if (this.gainNode) {
428
+ this.gainNode.gain.value = this.volume
429
+ }
430
+ }
431
+
432
+ destroy() {
433
+ if (this.audioContext) {
434
+ void this.audioContext.close()
435
+ this.audioContext = null
436
+ }
437
+ this.isInitialized = false
438
+ }
439
+ }
@@ -1,25 +1,50 @@
1
1
  /**
2
- * Gemini Veo Video Generation Provider Options
2
+ * Gemini Video Generation Provider Options
3
3
  *
4
- * Based on https://ai.google.dev/gemini-api/docs/video
4
+ * Covers two request paths behind the one video adapter:
5
+ * - Veo models — long-running operations via `:predictLongRunning`
6
+ * (https://ai.google.dev/gemini-api/docs/video)
7
+ * - Gemini Omni Flash — background jobs via the Interactions API
8
+ * (https://ai.google.dev/gemini-api/docs/omni)
5
9
  *
6
10
  * @experimental Video generation is an experimental feature and may change.
7
11
  */
12
+ import { GEMINI_INTERACTIONS_VIDEO_MODELS } from '../model-meta'
8
13
  import type { DurationOptions } from '@tanstack/ai/adapters'
9
- import type { GenerateVideosConfig } from '@google/genai'
14
+ import type { GenerateVideosConfig, Interactions } from '@google/genai'
10
15
  import type { GEMINI_VIDEO_MODELS } from '../model-meta'
11
16
 
12
17
  /**
13
- * Model type for Gemini Veo video generation.
18
+ * Model type for Gemini video generation (Veo + Omni Flash).
14
19
  * @experimental Video generation is an experimental feature and may change.
15
20
  */
16
21
  export type GeminiVideoModel = (typeof GEMINI_VIDEO_MODELS)[number]
17
22
 
18
23
  /**
19
- * Supported aspect ratios for Veo video generation. This is the `size` value
20
- * for the Gemini video adapter — Veo expresses output shape as an aspect
21
- * ratio (plus an optional `resolution` in `modelOptions`), not pixel
22
- * dimensions.
24
+ * Video models served by the Interactions API (Gemini Omni Flash) rather
25
+ * than Veo's `:predictLongRunning` operations flow.
26
+ * @experimental Omni video generation is an experimental feature and may change.
27
+ */
28
+ export type GeminiInteractionsVideoModel =
29
+ (typeof GEMINI_INTERACTIONS_VIDEO_MODELS)[number]
30
+
31
+ /**
32
+ * Runtime guard for the Interactions-served video models.
33
+ * @experimental Omni video generation is an experimental feature and may change.
34
+ */
35
+ export function isInteractionsVideoModel(
36
+ model: GeminiVideoModel,
37
+ ): model is GeminiInteractionsVideoModel {
38
+ return (GEMINI_INTERACTIONS_VIDEO_MODELS as ReadonlyArray<string>).includes(
39
+ model,
40
+ )
41
+ }
42
+
43
+ /**
44
+ * Supported aspect ratios for Gemini video generation. This is the `size`
45
+ * value for the Gemini video adapter — both Veo and Omni Flash express
46
+ * output shape as an aspect ratio (plus an optional `resolution` in Veo's
47
+ * `modelOptions`), not pixel dimensions.
23
48
  *
24
49
  * @experimental Video generation is an experimental feature and may change.
25
50
  */
@@ -49,13 +74,50 @@ export type GeminiVideoProviderOptions = Omit<
49
74
  | 'abortSignal'
50
75
  >
51
76
 
77
+ /**
78
+ * Provider-specific options for Gemini Omni Flash video generation on the
79
+ * Interactions API.
80
+ *
81
+ * Derived from the SDK's `Interactions.CreateModelInteractionParamsNonStreaming`,
82
+ * minus the fields the adapter manages itself:
83
+ * - `model` / `input` — set from the adapter's model and the `prompt`
84
+ * - `stream` / `background` — the adapter always creates a background job
85
+ * and polls it through the `generateVideo` jobs API
86
+ * - `response_modalities` / `response_format` — the adapter requests video
87
+ * output and maps the top-level `size` option onto
88
+ * `response_format.aspect_ratio`
89
+ * - `tools` / `response_mime_type` — not applicable to video generation
90
+ *
91
+ * Notable passthroughs:
92
+ * - `previous_interaction_id` — conversational video editing: chain a new
93
+ * prompt onto a prior Omni interaction to refine its video
94
+ * - `generation_config.video_config.task` — pin the task mode
95
+ * (`'text_to_video' | 'image_to_video' | 'reference_to_video' | 'edit'`)
96
+ * instead of letting the model infer it
97
+ *
98
+ * @experimental Omni video generation is an experimental feature and may change.
99
+ */
100
+ export type GeminiOmniVideoProviderOptions = Omit<
101
+ Interactions.CreateModelInteractionParamsNonStreaming,
102
+ | 'model'
103
+ | 'input'
104
+ | 'stream'
105
+ | 'background'
106
+ | 'response_modalities'
107
+ | 'response_format'
108
+ | 'response_mime_type'
109
+ | 'tools'
110
+ >
111
+
52
112
  /**
53
113
  * Model-specific provider options mapping.
54
114
  *
55
115
  * @experimental Video generation is an experimental feature and may change.
56
116
  */
57
117
  export type GeminiVideoModelProviderOptionsByName = {
58
- [TModel in GeminiVideoModel]: GeminiVideoProviderOptions
118
+ [TModel in GeminiVideoModel]: TModel extends GeminiInteractionsVideoModel
119
+ ? GeminiOmniVideoProviderOptions
120
+ : GeminiVideoProviderOptions
59
121
  }
60
122
 
61
123
  /**
@@ -70,17 +132,23 @@ export type GeminiVideoModelSizeByName = {
70
132
  /**
71
133
  * Per-model prompt input modalities. Every Veo model accepts image
72
134
  * conditioning inputs (first frame, last frame, reference images) alongside
73
- * the text prompt.
135
+ * the text prompt. Omni Flash additionally accepts video inputs (short
136
+ * reference clips / videos to edit).
74
137
  *
75
138
  * @experimental Video generation is an experimental feature and may change.
76
139
  */
77
140
  export type GeminiVideoModelInputModalitiesByName = {
78
- [TModel in GeminiVideoModel]: readonly ['image']
141
+ [TModel in GeminiVideoModel]: TModel extends GeminiInteractionsVideoModel
142
+ ? readonly ['image', 'video']
143
+ : readonly ['image']
79
144
  }
80
145
 
81
146
  /**
82
- * Per-model duration unions (seconds, as numbers — the API's
83
- * `parameters.durationSeconds` field is numeric).
147
+ * Per-model duration unions (seconds, as numbers — Veo's
148
+ * `parameters.durationSeconds` field is numeric; Omni Flash accepts a
149
+ * continuous 3–10 second range, fractional seconds included, so it stays
150
+ * `number` — the adapter rejects out-of-range values at job creation,
151
+ * against the range entry below).
84
152
  *
85
153
  * @experimental Video generation is an experimental feature and may change.
86
154
  */
@@ -88,15 +156,21 @@ export type GeminiVideoModelDurationByName = {
88
156
  'veo-3.1-generate-preview': 4 | 6 | 8
89
157
  'veo-3.1-fast-generate-preview': 4 | 6 | 8
90
158
  'veo-3.1-lite-generate-preview': 4 | 6 | 8
159
+ 'gemini-omni-flash-preview': number
91
160
  }
92
161
 
93
162
  /**
94
163
  * Runtime duration table backing `availableDurations()` / `snapDuration()`.
95
164
  *
96
- * Curated from the official Veo docs
165
+ * Veo values are curated from the official docs
97
166
  * (https://ai.google.dev/gemini-api/docs/video) — the Gemini OpenAPI spec
98
167
  * types the `:predictLongRunning` request's `parameters` as unconstrained,
99
168
  * so it carries no per-model duration information to derive these from.
169
+ * Omni Flash's 3–10s range was verified against the live API
170
+ * (2026-07-02): `response_format.duration` takes a `"<seconds>s"` string,
171
+ * fractional values are accepted, out-of-range values are rejected with
172
+ * "minimum allowed 3s" / "maximum allowed 10s", and omitting it defaults
173
+ * to a 10-second clip.
100
174
  *
101
175
  * @experimental Video generation is an experimental feature and may change.
102
176
  */
@@ -108,10 +182,16 @@ export const GEMINI_VIDEO_DURATIONS: {
108
182
  'veo-3.1-generate-preview': { kind: 'discrete', values: [4, 6, 8] },
109
183
  'veo-3.1-fast-generate-preview': { kind: 'discrete', values: [4, 6, 8] },
110
184
  'veo-3.1-lite-generate-preview': { kind: 'discrete', values: [4, 6, 8] },
185
+ 'gemini-omni-flash-preview': {
186
+ kind: 'range',
187
+ min: 3,
188
+ max: 10,
189
+ unit: 'seconds',
190
+ },
111
191
  }
112
192
 
113
193
  /**
114
- * Look up the duration options for a Veo model.
194
+ * Look up the duration options for a Gemini video model.
115
195
  *
116
196
  * @experimental Video generation is an experimental feature and may change.
117
197
  */