@tanstack/ai-gemini 0.19.1 → 0.20.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/audio.d.ts +1 -1
- package/dist/esm/adapters/audio.js.map +1 -1
- package/dist/esm/adapters/image.d.ts +1 -1
- package/dist/esm/adapters/image.js +17 -39
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/adapters/summarize.d.ts +1 -1
- package/dist/esm/adapters/summarize.js.map +1 -1
- package/dist/esm/adapters/text.d.ts +1 -1
- package/dist/esm/adapters/text.js.map +1 -1
- package/dist/esm/adapters/tts.d.ts +1 -1
- package/dist/esm/adapters/tts.js.map +1 -1
- package/dist/esm/adapters/video.d.ts +60 -11
- package/dist/esm/adapters/video.js +205 -6
- package/dist/esm/adapters/video.js.map +1 -1
- package/dist/esm/experimental/text-interactions/adapter.d.ts +1 -1
- package/dist/esm/experimental/text-interactions/adapter.js.map +1 -1
- package/dist/esm/index.d.ts +6 -3
- package/dist/esm/index.js +9 -3
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/model-meta.d.ts +11 -3
- package/dist/esm/model-meta.js +9 -1
- package/dist/esm/model-meta.js.map +1 -1
- package/dist/esm/realtime/adapter.d.ts +22 -0
- package/dist/esm/realtime/adapter.js +233 -0
- package/dist/esm/realtime/adapter.js.map +1 -0
- package/dist/esm/realtime/client.d.ts +98 -0
- package/dist/esm/realtime/client.js +389 -0
- package/dist/esm/realtime/client.js.map +1 -0
- package/dist/esm/realtime/index.d.ts +3 -0
- package/dist/esm/realtime/token.d.ts +26 -0
- package/dist/esm/realtime/token.js +39 -0
- package/dist/esm/realtime/token.js.map +1 -0
- package/dist/esm/realtime/types.d.ts +51 -0
- package/dist/esm/realtime/utils.d.ts +40 -0
- package/dist/esm/realtime/utils.js +350 -0
- package/dist/esm/realtime/utils.js.map +1 -0
- package/dist/esm/video/video-provider-options.d.ts +59 -14
- package/dist/esm/video/video-provider-options.js +15 -2
- package/dist/esm/video/video-provider-options.js.map +1 -1
- package/package.json +4 -4
- package/src/adapters/audio.ts +1 -1
- package/src/adapters/image.ts +25 -49
- package/src/adapters/summarize.ts +1 -1
- package/src/adapters/text.ts +1 -1
- package/src/adapters/tts.ts +1 -1
- package/src/adapters/video.ts +333 -16
- package/src/experimental/text-interactions/adapter.ts +2 -2
- package/src/index.ts +20 -2
- package/src/model-meta.ts +45 -2
- package/src/realtime/adapter.ts +311 -0
- package/src/realtime/client.ts +547 -0
- package/src/realtime/index.ts +14 -0
- package/src/realtime/token.ts +70 -0
- package/src/realtime/types.ts +94 -0
- package/src/realtime/utils.ts +439 -0
- package/src/video/video-provider-options.ts +95 -15
|
@@ -0,0 +1,350 @@
|
|
|
1
|
+
const captureWorkletCode = `
|
|
2
|
+
class AudioCaptureProcessor extends AudioWorkletProcessor {
|
|
3
|
+
constructor() {
|
|
4
|
+
super();
|
|
5
|
+
this.bufferSize = 512; // 32ms at 16kHz — per Gemini best practices (20-40ms chunks)
|
|
6
|
+
this.buffer = new Float32Array(this.bufferSize);
|
|
7
|
+
this.bufferIndex = 0;
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
process(inputs, outputs, parameters) {
|
|
11
|
+
const input = inputs[0];
|
|
12
|
+
|
|
13
|
+
if (input && input.length > 0) {
|
|
14
|
+
const inputChannel = input[0];
|
|
15
|
+
|
|
16
|
+
// Buffer the incoming audio
|
|
17
|
+
for (let i = 0; i < inputChannel.length; i++) {
|
|
18
|
+
this.buffer[this.bufferIndex++] = inputChannel[i];
|
|
19
|
+
|
|
20
|
+
// When buffer is full, send it to main thread
|
|
21
|
+
if (this.bufferIndex >= this.bufferSize) {
|
|
22
|
+
// Send the buffered audio to the main thread
|
|
23
|
+
this.port.postMessage({
|
|
24
|
+
type: "audio",
|
|
25
|
+
data: this.buffer.slice(),
|
|
26
|
+
});
|
|
27
|
+
|
|
28
|
+
// Reset buffer
|
|
29
|
+
this.bufferIndex = 0;
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
// Return true to keep the processor alive
|
|
35
|
+
return true;
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
// Register the processor
|
|
40
|
+
registerProcessor("audio-capture-processor", AudioCaptureProcessor);`;
|
|
41
|
+
const playbackWorkletCode = `
|
|
42
|
+
class PCMProcessor extends AudioWorkletProcessor {
|
|
43
|
+
constructor() {
|
|
44
|
+
super();
|
|
45
|
+
this.audioQueue = [];
|
|
46
|
+
this.currentOffset = 0; // Track position in current buffer (avoids slice())
|
|
47
|
+
|
|
48
|
+
this.port.onmessage = (event) => {
|
|
49
|
+
if (event.data === "interrupt") {
|
|
50
|
+
// Clear the queue on interrupt
|
|
51
|
+
this.audioQueue = [];
|
|
52
|
+
this.currentOffset = 0;
|
|
53
|
+
} else if (event.data instanceof Float32Array) {
|
|
54
|
+
// Add audio data to the queue
|
|
55
|
+
this.audioQueue.push(event.data);
|
|
56
|
+
}
|
|
57
|
+
};
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
process(inputs, outputs, parameters) {
|
|
61
|
+
const output = outputs[0];
|
|
62
|
+
if (output.length === 0) return true;
|
|
63
|
+
|
|
64
|
+
const channel = output[0];
|
|
65
|
+
let outputIndex = 0;
|
|
66
|
+
|
|
67
|
+
// Fill the output buffer from the queue
|
|
68
|
+
while (outputIndex < channel.length && this.audioQueue.length > 0) {
|
|
69
|
+
const currentBuffer = this.audioQueue[0];
|
|
70
|
+
|
|
71
|
+
if (!currentBuffer || currentBuffer.length === 0) {
|
|
72
|
+
this.audioQueue.shift();
|
|
73
|
+
this.currentOffset = 0;
|
|
74
|
+
continue;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
const remainingOutput = channel.length - outputIndex;
|
|
78
|
+
const remainingBuffer = currentBuffer.length - this.currentOffset;
|
|
79
|
+
const copyLength = Math.min(remainingOutput, remainingBuffer);
|
|
80
|
+
|
|
81
|
+
// Copy audio data to output using offset (no slice allocation)
|
|
82
|
+
for (let i = 0; i < copyLength; i++) {
|
|
83
|
+
channel[outputIndex++] = currentBuffer[this.currentOffset++];
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
// If we've consumed the entire buffer, move to the next one
|
|
87
|
+
if (this.currentOffset >= currentBuffer.length) {
|
|
88
|
+
this.audioQueue.shift();
|
|
89
|
+
this.currentOffset = 0;
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// Fill remaining output with silence
|
|
94
|
+
while (outputIndex < channel.length) {
|
|
95
|
+
channel[outputIndex++] = 0;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
return true;
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
registerProcessor("pcm-processor", PCMProcessor);`;
|
|
103
|
+
function calculateLevel(analyser) {
|
|
104
|
+
const data = new Uint8Array(analyser.fftSize);
|
|
105
|
+
analyser.getByteTimeDomainData(data);
|
|
106
|
+
let maxDeviation = 0;
|
|
107
|
+
for (const sample of data) {
|
|
108
|
+
const deviation = Math.abs(sample - 128);
|
|
109
|
+
if (deviation > maxDeviation) {
|
|
110
|
+
maxDeviation = deviation;
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
const normalized = maxDeviation / 128;
|
|
114
|
+
return Math.min(1, normalized * 1.5);
|
|
115
|
+
}
|
|
116
|
+
function base64ToArrayBuffer(base64) {
|
|
117
|
+
const binary = atob(base64);
|
|
118
|
+
const bytes = Uint8Array.from(binary, (char) => char.charCodeAt(0));
|
|
119
|
+
return bytes.buffer;
|
|
120
|
+
}
|
|
121
|
+
const emptyFrequencyData = new Uint8Array(1024);
|
|
122
|
+
const emptyTimeDomainData = new Uint8Array(2048).fill(128);
|
|
123
|
+
class AudioStreamer {
|
|
124
|
+
audioContext = null;
|
|
125
|
+
audioWorklet = null;
|
|
126
|
+
mediaStream = null;
|
|
127
|
+
analyser = null;
|
|
128
|
+
isStreaming = false;
|
|
129
|
+
sampleRate = 16e3;
|
|
130
|
+
client = null;
|
|
131
|
+
constructor(client) {
|
|
132
|
+
this.client = client;
|
|
133
|
+
}
|
|
134
|
+
get inputLevel() {
|
|
135
|
+
if (!this.analyser) return 0;
|
|
136
|
+
return calculateLevel(this.analyser);
|
|
137
|
+
}
|
|
138
|
+
get inputFrequencyData() {
|
|
139
|
+
if (!this.analyser) return emptyFrequencyData;
|
|
140
|
+
const data = new Uint8Array(this.analyser.frequencyBinCount);
|
|
141
|
+
this.analyser.getByteFrequencyData(data);
|
|
142
|
+
return data;
|
|
143
|
+
}
|
|
144
|
+
get inputTimeDomainData() {
|
|
145
|
+
if (!this.analyser) return emptyTimeDomainData;
|
|
146
|
+
const data = new Uint8Array(this.analyser.fftSize);
|
|
147
|
+
this.analyser.getByteTimeDomainData(data);
|
|
148
|
+
return data;
|
|
149
|
+
}
|
|
150
|
+
get inputSampleRate() {
|
|
151
|
+
return this.sampleRate;
|
|
152
|
+
}
|
|
153
|
+
async start() {
|
|
154
|
+
try {
|
|
155
|
+
const audioConstraints = {
|
|
156
|
+
sampleRate: this.sampleRate,
|
|
157
|
+
echoCancellation: true,
|
|
158
|
+
noiseSuppression: true,
|
|
159
|
+
autoGainControl: true
|
|
160
|
+
};
|
|
161
|
+
this.mediaStream = await navigator.mediaDevices.getUserMedia({
|
|
162
|
+
audio: audioConstraints
|
|
163
|
+
});
|
|
164
|
+
const track = this.mediaStream.getAudioTracks()[0];
|
|
165
|
+
const settings = track?.getSettings();
|
|
166
|
+
if (settings?.autoGainControl) {
|
|
167
|
+
console.warn("Native AGC not supported.");
|
|
168
|
+
}
|
|
169
|
+
this.audioContext = new AudioContext({
|
|
170
|
+
sampleRate: this.sampleRate
|
|
171
|
+
});
|
|
172
|
+
if (this.audioContext.state === "suspended") {
|
|
173
|
+
await this.audioContext.resume();
|
|
174
|
+
}
|
|
175
|
+
const workletBlob = new Blob([captureWorkletCode], {
|
|
176
|
+
type: "application/javascript"
|
|
177
|
+
});
|
|
178
|
+
const workletUrl = URL.createObjectURL(workletBlob);
|
|
179
|
+
await this.audioContext.audioWorklet.addModule(workletUrl);
|
|
180
|
+
URL.revokeObjectURL(workletUrl);
|
|
181
|
+
this.audioWorklet = new AudioWorkletNode(
|
|
182
|
+
this.audioContext,
|
|
183
|
+
"audio-capture-processor"
|
|
184
|
+
);
|
|
185
|
+
this.audioWorklet.port.onmessage = (event) => {
|
|
186
|
+
if (!this.isStreaming) return;
|
|
187
|
+
if (event.data.type === "audio") {
|
|
188
|
+
const inputData = event.data.data;
|
|
189
|
+
const pcmData = this.convertToPCM16(inputData);
|
|
190
|
+
const base64Audio = this.arrayBufferToBase64(pcmData);
|
|
191
|
+
if (this.client?.isSetupComplete) {
|
|
192
|
+
this.client.sendAudioMessage(base64Audio);
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
};
|
|
196
|
+
this.analyser = this.audioContext.createAnalyser();
|
|
197
|
+
this.analyser.fftSize = 2048;
|
|
198
|
+
this.analyser.smoothingTimeConstant = 0.3;
|
|
199
|
+
const source = this.audioContext.createMediaStreamSource(this.mediaStream);
|
|
200
|
+
source.connect(this.analyser);
|
|
201
|
+
this.analyser.connect(this.audioWorklet);
|
|
202
|
+
this.isStreaming = true;
|
|
203
|
+
} catch (error) {
|
|
204
|
+
this.stop();
|
|
205
|
+
throw error;
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
stop() {
|
|
209
|
+
this.isStreaming = false;
|
|
210
|
+
if (this.audioWorklet) {
|
|
211
|
+
this.audioWorklet.disconnect();
|
|
212
|
+
this.audioWorklet.port.close();
|
|
213
|
+
this.audioWorklet = null;
|
|
214
|
+
}
|
|
215
|
+
if (this.audioContext) {
|
|
216
|
+
void this.audioContext.close();
|
|
217
|
+
this.audioContext = null;
|
|
218
|
+
}
|
|
219
|
+
if (this.mediaStream) {
|
|
220
|
+
this.mediaStream.getTracks().forEach((track) => track.stop());
|
|
221
|
+
this.mediaStream = null;
|
|
222
|
+
}
|
|
223
|
+
}
|
|
224
|
+
startAudioCapture() {
|
|
225
|
+
if (this.mediaStream) {
|
|
226
|
+
for (const track of this.mediaStream.getAudioTracks()) {
|
|
227
|
+
track.enabled = true;
|
|
228
|
+
}
|
|
229
|
+
}
|
|
230
|
+
this.isStreaming = true;
|
|
231
|
+
}
|
|
232
|
+
stopAudioCapture() {
|
|
233
|
+
if (this.mediaStream) {
|
|
234
|
+
for (const track of this.mediaStream.getAudioTracks()) {
|
|
235
|
+
track.enabled = false;
|
|
236
|
+
}
|
|
237
|
+
}
|
|
238
|
+
this.isStreaming = false;
|
|
239
|
+
}
|
|
240
|
+
convertToPCM16(float32Array) {
|
|
241
|
+
const int16Array = new Int16Array(float32Array.length);
|
|
242
|
+
for (let i = 0; i < float32Array.length; i++) {
|
|
243
|
+
const sample = Math.max(-1, Math.min(1, float32Array[i] ?? 0));
|
|
244
|
+
int16Array[i] = sample * 32767;
|
|
245
|
+
}
|
|
246
|
+
return int16Array.buffer;
|
|
247
|
+
}
|
|
248
|
+
arrayBufferToBase64(buffer) {
|
|
249
|
+
const bytes = new Uint8Array(buffer);
|
|
250
|
+
const binary = String.fromCharCode(...bytes);
|
|
251
|
+
return btoa(binary);
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
class AudioPlayer {
|
|
255
|
+
audioContext = null;
|
|
256
|
+
workletNode = null;
|
|
257
|
+
gainNode = null;
|
|
258
|
+
analyser = null;
|
|
259
|
+
isInitialized = false;
|
|
260
|
+
volume = 1;
|
|
261
|
+
sampleRate = 24e3;
|
|
262
|
+
get outputLevel() {
|
|
263
|
+
if (!this.analyser) return 0;
|
|
264
|
+
return calculateLevel(this.analyser);
|
|
265
|
+
}
|
|
266
|
+
get outputFrequencyData() {
|
|
267
|
+
if (!this.analyser) return emptyFrequencyData;
|
|
268
|
+
const data = new Uint8Array(this.analyser.frequencyBinCount);
|
|
269
|
+
this.analyser.getByteFrequencyData(data);
|
|
270
|
+
return data;
|
|
271
|
+
}
|
|
272
|
+
get outputTimeDomainData() {
|
|
273
|
+
if (!this.analyser) return emptyTimeDomainData;
|
|
274
|
+
const data = new Uint8Array(this.analyser.fftSize);
|
|
275
|
+
this.analyser.getByteTimeDomainData(data);
|
|
276
|
+
return data;
|
|
277
|
+
}
|
|
278
|
+
get outputSampleRate() {
|
|
279
|
+
return this.sampleRate;
|
|
280
|
+
}
|
|
281
|
+
async init() {
|
|
282
|
+
if (this.isInitialized) return;
|
|
283
|
+
try {
|
|
284
|
+
this.audioContext = new AudioContext({
|
|
285
|
+
sampleRate: this.sampleRate
|
|
286
|
+
});
|
|
287
|
+
const workletBlob = new Blob([playbackWorkletCode], {
|
|
288
|
+
type: "application/javascript"
|
|
289
|
+
});
|
|
290
|
+
const workletUrl = URL.createObjectURL(workletBlob);
|
|
291
|
+
await this.audioContext.audioWorklet.addModule(workletUrl);
|
|
292
|
+
URL.revokeObjectURL(workletUrl);
|
|
293
|
+
this.workletNode = new AudioWorkletNode(
|
|
294
|
+
this.audioContext,
|
|
295
|
+
"pcm-processor"
|
|
296
|
+
);
|
|
297
|
+
this.gainNode = this.audioContext.createGain();
|
|
298
|
+
this.gainNode.gain.value = this.volume;
|
|
299
|
+
this.analyser = this.audioContext.createAnalyser();
|
|
300
|
+
this.analyser.fftSize = 2048;
|
|
301
|
+
this.analyser.smoothingTimeConstant = 0.3;
|
|
302
|
+
this.workletNode.connect(this.gainNode);
|
|
303
|
+
this.gainNode.connect(this.analyser);
|
|
304
|
+
this.analyser.connect(this.audioContext.destination);
|
|
305
|
+
this.isInitialized = true;
|
|
306
|
+
} catch (error) {
|
|
307
|
+
this.destroy();
|
|
308
|
+
throw error;
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
async play(pcmData) {
|
|
312
|
+
if (!this.isInitialized) {
|
|
313
|
+
await this.init();
|
|
314
|
+
}
|
|
315
|
+
if (this.audioContext?.state === "suspended") {
|
|
316
|
+
await this.audioContext.resume();
|
|
317
|
+
}
|
|
318
|
+
const inputArray = new Int16Array(pcmData);
|
|
319
|
+
const float32Data = new Float32Array(inputArray.length);
|
|
320
|
+
for (let i = 0; i < inputArray.length; i++) {
|
|
321
|
+
float32Data[i] = (inputArray[i] ?? 0) / 32768;
|
|
322
|
+
}
|
|
323
|
+
this.workletNode?.port.postMessage(float32Data);
|
|
324
|
+
}
|
|
325
|
+
/* Interrupt playback */
|
|
326
|
+
interrupt() {
|
|
327
|
+
if (this.workletNode) {
|
|
328
|
+
this.workletNode.port.postMessage("interrupt");
|
|
329
|
+
}
|
|
330
|
+
}
|
|
331
|
+
setVolume(volume) {
|
|
332
|
+
this.volume = Math.max(0, Math.min(1, volume));
|
|
333
|
+
if (this.gainNode) {
|
|
334
|
+
this.gainNode.gain.value = this.volume;
|
|
335
|
+
}
|
|
336
|
+
}
|
|
337
|
+
destroy() {
|
|
338
|
+
if (this.audioContext) {
|
|
339
|
+
void this.audioContext.close();
|
|
340
|
+
this.audioContext = null;
|
|
341
|
+
}
|
|
342
|
+
this.isInitialized = false;
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
export {
|
|
346
|
+
AudioPlayer,
|
|
347
|
+
AudioStreamer,
|
|
348
|
+
base64ToArrayBuffer
|
|
349
|
+
};
|
|
350
|
+
//# sourceMappingURL=utils.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"utils.js","sources":["../../../src/realtime/utils.ts"],"sourcesContent":["import type { GeminiLiveClient } from './client'\n\n/**\n * Audio Worklet Processor for capturing and processing audio\n */\nconst captureWorkletCode = `\nclass AudioCaptureProcessor extends AudioWorkletProcessor {\n constructor() {\n super();\n this.bufferSize = 512; // 32ms at 16kHz — per Gemini best practices (20-40ms chunks)\n this.buffer = new Float32Array(this.bufferSize);\n this.bufferIndex = 0;\n }\n\n process(inputs, outputs, parameters) {\n const input = inputs[0];\n\n if (input && input.length > 0) {\n const inputChannel = input[0];\n\n // Buffer the incoming audio\n for (let i = 0; i < inputChannel.length; i++) {\n this.buffer[this.bufferIndex++] = inputChannel[i];\n\n // When buffer is full, send it to main thread\n if (this.bufferIndex >= this.bufferSize) {\n // Send the buffered audio to the main thread\n this.port.postMessage({\n type: \"audio\",\n data: this.buffer.slice(),\n });\n\n // Reset buffer\n this.bufferIndex = 0;\n }\n }\n }\n\n // Return true to keep the processor alive\n return true;\n }\n}\n\n// Register the processor\nregisterProcessor(\"audio-capture-processor\", AudioCaptureProcessor);`\n\n/**\n * Audio Playback Worklet Processor for playing PCM audio.\n * Uses an offset tracker instead of slice() to avoid allocations\n * on the real-time audio thread.\n */\nconst playbackWorkletCode = `\nclass PCMProcessor extends AudioWorkletProcessor {\n constructor() {\n super();\n this.audioQueue = [];\n this.currentOffset = 0; // Track position in current buffer (avoids slice())\n\n this.port.onmessage = (event) => {\n if (event.data === \"interrupt\") {\n // Clear the queue on interrupt\n this.audioQueue = [];\n this.currentOffset = 0;\n } else if (event.data instanceof Float32Array) {\n // Add audio data to the queue\n this.audioQueue.push(event.data);\n }\n };\n }\n\n process(inputs, outputs, parameters) {\n const output = outputs[0];\n if (output.length === 0) return true;\n\n const channel = output[0];\n let outputIndex = 0;\n\n // Fill the output buffer from the queue\n while (outputIndex < channel.length && this.audioQueue.length > 0) {\n const currentBuffer = this.audioQueue[0];\n\n if (!currentBuffer || currentBuffer.length === 0) {\n this.audioQueue.shift();\n this.currentOffset = 0;\n continue;\n }\n\n const remainingOutput = channel.length - outputIndex;\n const remainingBuffer = currentBuffer.length - this.currentOffset;\n const copyLength = Math.min(remainingOutput, remainingBuffer);\n\n // Copy audio data to output using offset (no slice allocation)\n for (let i = 0; i < copyLength; i++) {\n channel[outputIndex++] = currentBuffer[this.currentOffset++];\n }\n\n // If we've consumed the entire buffer, move to the next one\n if (this.currentOffset >= currentBuffer.length) {\n this.audioQueue.shift();\n this.currentOffset = 0;\n }\n }\n\n // Fill remaining output with silence\n while (outputIndex < channel.length) {\n channel[outputIndex++] = 0;\n }\n\n return true;\n }\n}\n\nregisterProcessor(\"pcm-processor\", PCMProcessor);`\n\nfunction calculateLevel(analyser: AnalyserNode): number {\n const data = new Uint8Array(analyser.fftSize)\n analyser.getByteTimeDomainData(data)\n\n // Find peak deviation from center (128 is silence)\n // This is more responsive than RMS for voice level meters\n let maxDeviation = 0\n for (const sample of data) {\n const deviation = Math.abs(sample - 128)\n if (deviation > maxDeviation) {\n maxDeviation = deviation\n }\n }\n\n // Normalize to 0-1 range (max deviation is 128)\n // Scale by 1.5x so that ~66% amplitude reads as full scale\n // This provides good visual feedback without pegging too early\n const normalized = maxDeviation / 128\n return Math.min(1, normalized * 1.5)\n}\n\nexport function base64ToArrayBuffer(base64: string): ArrayBuffer {\n const binary = atob(base64)\n const bytes = Uint8Array.from(binary, (char) => char.charCodeAt(0))\n return bytes.buffer\n}\n\n// Empty arrays for when visualization isn't available\n// frequencyBinCount = fftSize / 2 = 1024\nconst emptyFrequencyData = new Uint8Array(1024)\nconst emptyTimeDomainData = new Uint8Array(2048).fill(128) // 128 is silence\n\nexport class AudioStreamer {\n private audioContext: AudioContext | null = null\n private audioWorklet: AudioWorkletNode | null = null\n private mediaStream: MediaStream | null = null\n private analyser: AnalyserNode | null = null\n private isStreaming = false\n private readonly sampleRate = 16000\n private readonly client: GeminiLiveClient | null = null\n\n constructor(client: GeminiLiveClient) {\n this.client = client\n }\n\n get inputLevel() {\n if (!this.analyser) return 0\n return calculateLevel(this.analyser)\n }\n\n get inputFrequencyData() {\n if (!this.analyser) return emptyFrequencyData\n const data = new Uint8Array(this.analyser.frequencyBinCount)\n this.analyser.getByteFrequencyData(data)\n return data\n }\n\n get inputTimeDomainData() {\n if (!this.analyser) return emptyTimeDomainData\n const data = new Uint8Array(this.analyser.fftSize)\n this.analyser.getByteTimeDomainData(data)\n return data\n }\n\n get inputSampleRate() {\n return this.sampleRate\n }\n\n async start() {\n try {\n const audioConstraints: MediaTrackConstraints = {\n sampleRate: this.sampleRate,\n echoCancellation: true,\n noiseSuppression: true,\n autoGainControl: true,\n }\n\n // Get microphone access\n this.mediaStream = await navigator.mediaDevices.getUserMedia({\n audio: audioConstraints,\n })\n\n // Check if native AGC is active\n const track = this.mediaStream.getAudioTracks()[0]\n const settings = track?.getSettings()\n\n if (settings?.autoGainControl) {\n console.warn('Native AGC not supported.')\n }\n\n // Create audio context\n this.audioContext = new AudioContext({\n sampleRate: this.sampleRate,\n })\n\n if (this.audioContext.state === 'suspended') {\n await this.audioContext.resume()\n }\n\n const workletBlob = new Blob([captureWorkletCode], {\n type: 'application/javascript',\n })\n const workletUrl = URL.createObjectURL(workletBlob)\n\n // Load the audio worklet module, then release the blob URL.\n await this.audioContext.audioWorklet.addModule(workletUrl)\n URL.revokeObjectURL(workletUrl)\n\n // Create the audio worklet node\n this.audioWorklet = new AudioWorkletNode(\n this.audioContext,\n 'audio-capture-processor',\n )\n\n // Set up message handling from the worklet\n this.audioWorklet.port.onmessage = (event) => {\n if (!this.isStreaming) return\n\n if (event.data.type === 'audio') {\n const inputData = event.data.data\n const pcmData = this.convertToPCM16(inputData)\n const base64Audio = this.arrayBufferToBase64(pcmData)\n\n // Send to Gemini only if after setup complete\n if (this.client?.isSetupComplete) {\n this.client.sendAudioMessage(base64Audio)\n }\n }\n }\n\n // Create analyser for volume detection\n this.analyser = this.audioContext.createAnalyser()\n this.analyser.fftSize = 2048 // Larger size for more accurate level detection\n this.analyser.smoothingTimeConstant = 0.3\n\n // Connect the audio graph\n const source = this.audioContext.createMediaStreamSource(this.mediaStream)\n source.connect(this.analyser)\n this.analyser.connect(this.audioWorklet)\n\n // Start streaming\n this.isStreaming = true\n } catch (error) {\n // Clean up the mic + audio context if setup failed partway through.\n this.stop()\n throw error\n }\n }\n\n stop() {\n this.isStreaming = false\n\n if (this.audioWorklet) {\n this.audioWorklet.disconnect()\n this.audioWorklet.port.close()\n this.audioWorklet = null\n }\n\n if (this.audioContext) {\n void this.audioContext.close()\n this.audioContext = null\n }\n\n if (this.mediaStream) {\n this.mediaStream.getTracks().forEach((track) => track.stop())\n this.mediaStream = null\n }\n }\n\n startAudioCapture() {\n if (this.mediaStream) {\n for (const track of this.mediaStream.getAudioTracks()) {\n track.enabled = true\n }\n }\n this.isStreaming = true\n }\n\n stopAudioCapture() {\n if (this.mediaStream) {\n // Disable tracks rather than stopping them to allow re-enabling\n for (const track of this.mediaStream.getAudioTracks()) {\n track.enabled = false\n }\n }\n this.isStreaming = false\n }\n\n private convertToPCM16(float32Array: Float32Array): ArrayBuffer {\n const int16Array = new Int16Array(float32Array.length)\n for (let i = 0; i < float32Array.length; i++) {\n const sample = Math.max(-1, Math.min(1, float32Array[i] ?? 0))\n int16Array[i] = sample * 0x7fff\n }\n return int16Array.buffer\n }\n\n private arrayBufferToBase64(buffer: ArrayBuffer): string {\n const bytes = new Uint8Array(buffer)\n const binary = String.fromCharCode(...bytes)\n return btoa(binary)\n }\n}\n\nexport class AudioPlayer {\n private audioContext: AudioContext | null = null\n private workletNode: AudioWorkletNode | null = null\n private gainNode: GainNode | null = null\n private analyser: AnalyserNode | null = null\n private isInitialized = false\n private volume = 1.0\n private readonly sampleRate = 24000\n\n get outputLevel() {\n if (!this.analyser) return 0\n return calculateLevel(this.analyser)\n }\n\n get outputFrequencyData() {\n if (!this.analyser) return emptyFrequencyData\n const data = new Uint8Array(this.analyser.frequencyBinCount)\n this.analyser.getByteFrequencyData(data)\n return data\n }\n\n get outputTimeDomainData() {\n if (!this.analyser) return emptyTimeDomainData\n const data = new Uint8Array(this.analyser.fftSize)\n this.analyser.getByteTimeDomainData(data)\n return data\n }\n\n get outputSampleRate() {\n return this.sampleRate\n }\n\n async init() {\n if (this.isInitialized) return\n\n try {\n // Create audio context at 24kHz to match Gemini\n this.audioContext = new AudioContext({\n sampleRate: this.sampleRate,\n })\n\n const workletBlob = new Blob([playbackWorkletCode], {\n type: 'application/javascript',\n })\n const workletUrl = URL.createObjectURL(workletBlob)\n\n // Load the audio worklet module, then release the blob URL.\n await this.audioContext.audioWorklet.addModule(workletUrl)\n URL.revokeObjectURL(workletUrl)\n\n // Create worklet node\n this.workletNode = new AudioWorkletNode(\n this.audioContext,\n 'pcm-processor',\n )\n\n // Create gain node for volume control\n this.gainNode = this.audioContext.createGain()\n this.gainNode.gain.value = this.volume\n\n // Create analyser for volume detection\n this.analyser = this.audioContext.createAnalyser()\n this.analyser.fftSize = 2048 // Larger size for more accurate level detection\n this.analyser.smoothingTimeConstant = 0.3\n\n // Connect nodes\n this.workletNode.connect(this.gainNode)\n this.gainNode.connect(this.analyser)\n this.analyser.connect(this.audioContext.destination)\n\n this.isInitialized = true\n } catch (error) {\n // Release the audio context if initialization failed partway through.\n this.destroy()\n throw error\n }\n }\n\n async play(pcmData: ArrayBuffer) {\n if (!this.isInitialized) {\n await this.init()\n }\n\n // Resume audio context if suspended\n if (this.audioContext?.state === 'suspended') {\n await this.audioContext.resume()\n }\n\n // Convert PCM16 LE to Float32\n const inputArray = new Int16Array(pcmData)\n const float32Data = new Float32Array(inputArray.length)\n for (let i = 0; i < inputArray.length; i++) {\n float32Data[i] = (inputArray[i] ?? 0) / 32768\n }\n\n // Send to worklet for playback\n this.workletNode?.port.postMessage(float32Data)\n }\n\n /* Interrupt playback */\n interrupt() {\n if (this.workletNode) {\n this.workletNode.port.postMessage('interrupt')\n }\n }\n\n setVolume(volume: number) {\n this.volume = Math.max(0, Math.min(1, volume))\n if (this.gainNode) {\n this.gainNode.gain.value = this.volume\n }\n }\n\n destroy() {\n if (this.audioContext) {\n void this.audioContext.close()\n this.audioContext = null\n }\n this.isInitialized = false\n }\n}\n"],"names":[],"mappings":"AAKA,MAAM,qBAAqB;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AA8C3B,MAAM,sBAAsB;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AA+D5B,SAAS,eAAe,UAAgC;AACtD,QAAM,OAAO,IAAI,WAAW,SAAS,OAAO;AAC5C,WAAS,sBAAsB,IAAI;AAInC,MAAI,eAAe;AACnB,aAAW,UAAU,MAAM;AACzB,UAAM,YAAY,KAAK,IAAI,SAAS,GAAG;AACvC,QAAI,YAAY,cAAc;AAC5B,qBAAe;AAAA,IACjB;AAAA,EACF;AAKA,QAAM,aAAa,eAAe;AAClC,SAAO,KAAK,IAAI,GAAG,aAAa,GAAG;AACrC;AAEO,SAAS,oBAAoB,QAA6B;AAC/D,QAAM,SAAS,KAAK,MAAM;AAC1B,QAAM,QAAQ,WAAW,KAAK,QAAQ,CAAC,SAAS,KAAK,WAAW,CAAC,CAAC;AAClE,SAAO,MAAM;AACf;AAIA,MAAM,qBAAqB,IAAI,WAAW,IAAI;AAC9C,MAAM,sBAAsB,IAAI,WAAW,IAAI,EAAE,KAAK,GAAG;AAElD,MAAM,cAAc;AAAA,EACjB,eAAoC;AAAA,EACpC,eAAwC;AAAA,EACxC,cAAkC;AAAA,EAClC,WAAgC;AAAA,EAChC,cAAc;AAAA,EACL,aAAa;AAAA,EACb,SAAkC;AAAA,EAEnD,YAAY,QAA0B;AACpC,SAAK,SAAS;AAAA,EAChB;AAAA,EAEA,IAAI,aAAa;AACf,QAAI,CAAC,KAAK,SAAU,QAAO;AAC3B,WAAO,eAAe,KAAK,QAAQ;AAAA,EACrC;AAAA,EAEA,IAAI,qBAAqB;AACvB,QAAI,CAAC,KAAK,SAAU,QAAO;AAC3B,UAAM,OAAO,IAAI,WAAW,KAAK,SAAS,iBAAiB;AAC3D,SAAK,SAAS,qBAAqB,IAAI;AACvC,WAAO;AAAA,EACT;AAAA,EAEA,IAAI,sBAAsB;AACxB,QAAI,CAAC,KAAK,SAAU,QAAO;AAC3B,UAAM,OAAO,IAAI,WAAW,KAAK,SAAS,OAAO;AACjD,SAAK,SAAS,sBAAsB,IAAI;AACxC,WAAO;AAAA,EACT;AAAA,EAEA,IAAI,kBAAkB;AACpB,WAAO,KAAK;AAAA,EACd;AAAA,EAEA,MAAM,QAAQ;AACZ,QAAI;AACF,YAAM,mBAA0C;AAAA,QAC9C,YAAY,KAAK;AAAA,QACjB,kBAAkB;AAAA,QAClB,kBAAkB;AAAA,QAClB,iBAAiB;AAAA,MAAA;AAInB,WAAK,cAAc,MAAM,UAAU,aAAa,aAAa;AAAA,QAC3D,OAAO;AAAA,MAAA,CACR;AAGD,YAAM,QAAQ,KAAK,YAAY,eAAA,EAAiB,CAAC;AACjD,YAAM,WAAW,OAAO,YAAA;AAExB,UAAI,UAAU,iBAAiB;AAC7B,gBAAQ,KAAK,2BAA2B;AAAA,MAC1C;AAGA,WAAK,eAAe,IAAI,aAAa;AAAA,QACnC,YAAY,KAAK;AAAA,MAAA,CAClB;AAED,UAAI,KAAK,aAAa,UAAU,aAAa;AAC3C,cAAM,KAAK,aAAa,OAAA;AAAA,MAC1B;AAEA,YAAM,cAAc,IAAI,KAAK,CAAC,kBAAkB,GAAG;AAAA,QACjD,MAAM;AAAA,MAAA,CACP;AACD,YAAM,aAAa,IAAI,gBAAgB,WAAW;AAGlD,YAAM,KAAK,aAAa,aAAa,UAAU,UAAU;AACzD,UAAI,gBAAgB,UAAU;AAG9B,WAAK,eAAe,IAAI;AAAA,QACtB,KAAK;AAAA,QACL;AAAA,MAAA;AAIF,WAAK,aAAa,KAAK,YAAY,CAAC,UAAU;AAC5C,YAAI,CAAC,KAAK,YAAa;AAEvB,YAAI,MAAM,KAAK,SAAS,SAAS;AAC/B,gBAAM,YAAY,MAAM,KAAK;AAC7B,gBAAM,UAAU,KAAK,eAAe,SAAS;AAC7C,gBAAM,cAAc,KAAK,oBAAoB,OAAO;AAGpD,cAAI,KAAK,QAAQ,iBAAiB;AAChC,iBAAK,OAAO,iBAAiB,WAAW;AAAA,UAC1C;AAAA,QACF;AAAA,MACF;AAGA,WAAK,WAAW,KAAK,aAAa,eAAA;AAClC,WAAK,SAAS,UAAU;AACxB,WAAK,SAAS,wBAAwB;AAGtC,YAAM,SAAS,KAAK,aAAa,wBAAwB,KAAK,WAAW;AACzE,aAAO,QAAQ,KAAK,QAAQ;AAC5B,WAAK,SAAS,QAAQ,KAAK,YAAY;AAGvC,WAAK,cAAc;AAAA,IACrB,SAAS,OAAO;AAEd,WAAK,KAAA;AACL,YAAM;AAAA,IACR;AAAA,EACF;AAAA,EAEA,OAAO;AACL,SAAK,cAAc;AAEnB,QAAI,KAAK,cAAc;AACrB,WAAK,aAAa,WAAA;AAClB,WAAK,aAAa,KAAK,MAAA;AACvB,WAAK,eAAe;AAAA,IACtB;AAEA,QAAI,KAAK,cAAc;AACrB,WAAK,KAAK,aAAa,MAAA;AACvB,WAAK,eAAe;AAAA,IACtB;AAEA,QAAI,KAAK,aAAa;AACpB,WAAK,YAAY,YAAY,QAAQ,CAAC,UAAU,MAAM,MAAM;AAC5D,WAAK,cAAc;AAAA,IACrB;AAAA,EACF;AAAA,EAEA,oBAAoB;AAClB,QAAI,KAAK,aAAa;AACpB,iBAAW,SAAS,KAAK,YAAY,eAAA,GAAkB;AACrD,cAAM,UAAU;AAAA,MAClB;AAAA,IACF;AACA,SAAK,cAAc;AAAA,EACrB;AAAA,EAEA,mBAAmB;AACjB,QAAI,KAAK,aAAa;AAEpB,iBAAW,SAAS,KAAK,YAAY,eAAA,GAAkB;AACrD,cAAM,UAAU;AAAA,MAClB;AAAA,IACF;AACA,SAAK,cAAc;AAAA,EACrB;AAAA,EAEQ,eAAe,cAAyC;AAC9D,UAAM,aAAa,IAAI,WAAW,aAAa,MAAM;AACrD,aAAS,IAAI,GAAG,IAAI,aAAa,QAAQ,KAAK;AAC5C,YAAM,SAAS,KAAK,IAAI,IAAI,KAAK,IAAI,GAAG,aAAa,CAAC,KAAK,CAAC,CAAC;AAC7D,iBAAW,CAAC,IAAI,SAAS;AAAA,IAC3B;AACA,WAAO,WAAW;AAAA,EACpB;AAAA,EAEQ,oBAAoB,QAA6B;AACvD,UAAM,QAAQ,IAAI,WAAW,MAAM;AACnC,UAAM,SAAS,OAAO,aAAa,GAAG,KAAK;AAC3C,WAAO,KAAK,MAAM;AAAA,EACpB;AACF;AAEO,MAAM,YAAY;AAAA,EACf,eAAoC;AAAA,EACpC,cAAuC;AAAA,EACvC,WAA4B;AAAA,EAC5B,WAAgC;AAAA,EAChC,gBAAgB;AAAA,EAChB,SAAS;AAAA,EACA,aAAa;AAAA,EAE9B,IAAI,cAAc;AAChB,QAAI,CAAC,KAAK,SAAU,QAAO;AAC3B,WAAO,eAAe,KAAK,QAAQ;AAAA,EACrC;AAAA,EAEA,IAAI,sBAAsB;AACxB,QAAI,CAAC,KAAK,SAAU,QAAO;AAC3B,UAAM,OAAO,IAAI,WAAW,KAAK,SAAS,iBAAiB;AAC3D,SAAK,SAAS,qBAAqB,IAAI;AACvC,WAAO;AAAA,EACT;AAAA,EAEA,IAAI,uBAAuB;AACzB,QAAI,CAAC,KAAK,SAAU,QAAO;AAC3B,UAAM,OAAO,IAAI,WAAW,KAAK,SAAS,OAAO;AACjD,SAAK,SAAS,sBAAsB,IAAI;AACxC,WAAO;AAAA,EACT;AAAA,EAEA,IAAI,mBAAmB;AACrB,WAAO,KAAK;AAAA,EACd;AAAA,EAEA,MAAM,OAAO;AACX,QAAI,KAAK,cAAe;AAExB,QAAI;AAEF,WAAK,eAAe,IAAI,aAAa;AAAA,QACnC,YAAY,KAAK;AAAA,MAAA,CAClB;AAED,YAAM,cAAc,IAAI,KAAK,CAAC,mBAAmB,GAAG;AAAA,QAClD,MAAM;AAAA,MAAA,CACP;AACD,YAAM,aAAa,IAAI,gBAAgB,WAAW;AAGlD,YAAM,KAAK,aAAa,aAAa,UAAU,UAAU;AACzD,UAAI,gBAAgB,UAAU;AAG9B,WAAK,cAAc,IAAI;AAAA,QACrB,KAAK;AAAA,QACL;AAAA,MAAA;AAIF,WAAK,WAAW,KAAK,aAAa,WAAA;AAClC,WAAK,SAAS,KAAK,QAAQ,KAAK;AAGhC,WAAK,WAAW,KAAK,aAAa,eAAA;AAClC,WAAK,SAAS,UAAU;AACxB,WAAK,SAAS,wBAAwB;AAGtC,WAAK,YAAY,QAAQ,KAAK,QAAQ;AACtC,WAAK,SAAS,QAAQ,KAAK,QAAQ;AACnC,WAAK,SAAS,QAAQ,KAAK,aAAa,WAAW;AAEnD,WAAK,gBAAgB;AAAA,IACvB,SAAS,OAAO;AAEd,WAAK,QAAA;AACL,YAAM;AAAA,IACR;AAAA,EACF;AAAA,EAEA,MAAM,KAAK,SAAsB;AAC/B,QAAI,CAAC,KAAK,eAAe;AACvB,YAAM,KAAK,KAAA;AAAA,IACb;AAGA,QAAI,KAAK,cAAc,UAAU,aAAa;AAC5C,YAAM,KAAK,aAAa,OAAA;AAAA,IAC1B;AAGA,UAAM,aAAa,IAAI,WAAW,OAAO;AACzC,UAAM,cAAc,IAAI,aAAa,WAAW,MAAM;AACtD,aAAS,IAAI,GAAG,IAAI,WAAW,QAAQ,KAAK;AAC1C,kBAAY,CAAC,KAAK,WAAW,CAAC,KAAK,KAAK;AAAA,IAC1C;AAGA,SAAK,aAAa,KAAK,YAAY,WAAW;AAAA,EAChD;AAAA;AAAA,EAGA,YAAY;AACV,QAAI,KAAK,aAAa;AACpB,WAAK,YAAY,KAAK,YAAY,WAAW;AAAA,IAC/C;AAAA,EACF;AAAA,EAEA,UAAU,QAAgB;AACxB,SAAK,SAAS,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,MAAM,CAAC;AAC7C,QAAI,KAAK,UAAU;AACjB,WAAK,SAAS,KAAK,QAAQ,KAAK;AAAA,IAClC;AAAA,EACF;AAAA,EAEA,UAAU;AACR,QAAI,KAAK,cAAc;AACrB,WAAK,KAAK,aAAa,MAAA;AACvB,WAAK,eAAe;AAAA,IACtB;AACA,SAAK,gBAAgB;AAAA,EACvB;AACF;"}
|
|
@@ -1,16 +1,27 @@
|
|
|
1
|
+
import { GEMINI_INTERACTIONS_VIDEO_MODELS, GEMINI_VIDEO_MODELS } from '../model-meta.js';
|
|
1
2
|
import { DurationOptions } from '@tanstack/ai/adapters';
|
|
2
|
-
import { GenerateVideosConfig } from '@google/genai';
|
|
3
|
-
import { GEMINI_VIDEO_MODELS } from '../model-meta.js';
|
|
3
|
+
import { GenerateVideosConfig, Interactions } from '@google/genai';
|
|
4
4
|
/**
|
|
5
|
-
* Model type for Gemini Veo
|
|
5
|
+
* Model type for Gemini video generation (Veo + Omni Flash).
|
|
6
6
|
* @experimental Video generation is an experimental feature and may change.
|
|
7
7
|
*/
|
|
8
8
|
export type GeminiVideoModel = (typeof GEMINI_VIDEO_MODELS)[number];
|
|
9
9
|
/**
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
|
|
10
|
+
* Video models served by the Interactions API (Gemini Omni Flash) rather
|
|
11
|
+
* than Veo's `:predictLongRunning` operations flow.
|
|
12
|
+
* @experimental Omni video generation is an experimental feature and may change.
|
|
13
|
+
*/
|
|
14
|
+
export type GeminiInteractionsVideoModel = (typeof GEMINI_INTERACTIONS_VIDEO_MODELS)[number];
|
|
15
|
+
/**
|
|
16
|
+
* Runtime guard for the Interactions-served video models.
|
|
17
|
+
* @experimental Omni video generation is an experimental feature and may change.
|
|
18
|
+
*/
|
|
19
|
+
export declare function isInteractionsVideoModel(model: GeminiVideoModel): model is GeminiInteractionsVideoModel;
|
|
20
|
+
/**
|
|
21
|
+
* Supported aspect ratios for Gemini video generation. This is the `size`
|
|
22
|
+
* value for the Gemini video adapter — both Veo and Omni Flash express
|
|
23
|
+
* output shape as an aspect ratio (plus an optional `resolution` in Veo's
|
|
24
|
+
* `modelOptions`), not pixel dimensions.
|
|
14
25
|
*
|
|
15
26
|
* @experimental Video generation is an experimental feature and may change.
|
|
16
27
|
*/
|
|
@@ -30,13 +41,37 @@ export type GeminiVideoSize = '16:9' | '9:16';
|
|
|
30
41
|
* @experimental Video generation is an experimental feature and may change.
|
|
31
42
|
*/
|
|
32
43
|
export type GeminiVideoProviderOptions = Omit<GenerateVideosConfig, 'durationSeconds' | 'aspectRatio' | 'lastFrame' | 'referenceImages' | 'httpOptions' | 'abortSignal'>;
|
|
44
|
+
/**
|
|
45
|
+
* Provider-specific options for Gemini Omni Flash video generation on the
|
|
46
|
+
* Interactions API.
|
|
47
|
+
*
|
|
48
|
+
* Derived from the SDK's `Interactions.CreateModelInteractionParamsNonStreaming`,
|
|
49
|
+
* minus the fields the adapter manages itself:
|
|
50
|
+
* - `model` / `input` — set from the adapter's model and the `prompt`
|
|
51
|
+
* - `stream` / `background` — the adapter always creates a background job
|
|
52
|
+
* and polls it through the `generateVideo` jobs API
|
|
53
|
+
* - `response_modalities` / `response_format` — the adapter requests video
|
|
54
|
+
* output and maps the top-level `size` option onto
|
|
55
|
+
* `response_format.aspect_ratio`
|
|
56
|
+
* - `tools` / `response_mime_type` — not applicable to video generation
|
|
57
|
+
*
|
|
58
|
+
* Notable passthroughs:
|
|
59
|
+
* - `previous_interaction_id` — conversational video editing: chain a new
|
|
60
|
+
* prompt onto a prior Omni interaction to refine its video
|
|
61
|
+
* - `generation_config.video_config.task` — pin the task mode
|
|
62
|
+
* (`'text_to_video' | 'image_to_video' | 'reference_to_video' | 'edit'`)
|
|
63
|
+
* instead of letting the model infer it
|
|
64
|
+
*
|
|
65
|
+
* @experimental Omni video generation is an experimental feature and may change.
|
|
66
|
+
*/
|
|
67
|
+
export type GeminiOmniVideoProviderOptions = Omit<Interactions.CreateModelInteractionParamsNonStreaming, 'model' | 'input' | 'stream' | 'background' | 'response_modalities' | 'response_format' | 'response_mime_type' | 'tools'>;
|
|
33
68
|
/**
|
|
34
69
|
* Model-specific provider options mapping.
|
|
35
70
|
*
|
|
36
71
|
* @experimental Video generation is an experimental feature and may change.
|
|
37
72
|
*/
|
|
38
73
|
export type GeminiVideoModelProviderOptionsByName = {
|
|
39
|
-
[TModel in GeminiVideoModel]: GeminiVideoProviderOptions;
|
|
74
|
+
[TModel in GeminiVideoModel]: TModel extends GeminiInteractionsVideoModel ? GeminiOmniVideoProviderOptions : GeminiVideoProviderOptions;
|
|
40
75
|
};
|
|
41
76
|
/**
|
|
42
77
|
* Model-specific size (aspect ratio) mapping.
|
|
@@ -49,16 +84,20 @@ export type GeminiVideoModelSizeByName = {
|
|
|
49
84
|
/**
|
|
50
85
|
* Per-model prompt input modalities. Every Veo model accepts image
|
|
51
86
|
* conditioning inputs (first frame, last frame, reference images) alongside
|
|
52
|
-
* the text prompt.
|
|
87
|
+
* the text prompt. Omni Flash additionally accepts video inputs (short
|
|
88
|
+
* reference clips / videos to edit).
|
|
53
89
|
*
|
|
54
90
|
* @experimental Video generation is an experimental feature and may change.
|
|
55
91
|
*/
|
|
56
92
|
export type GeminiVideoModelInputModalitiesByName = {
|
|
57
|
-
[TModel in GeminiVideoModel]: readonly ['image'];
|
|
93
|
+
[TModel in GeminiVideoModel]: TModel extends GeminiInteractionsVideoModel ? readonly ['image', 'video'] : readonly ['image'];
|
|
58
94
|
};
|
|
59
95
|
/**
|
|
60
|
-
* Per-model duration unions (seconds, as numbers —
|
|
61
|
-
* `parameters.durationSeconds` field is numeric
|
|
96
|
+
* Per-model duration unions (seconds, as numbers — Veo's
|
|
97
|
+
* `parameters.durationSeconds` field is numeric; Omni Flash accepts a
|
|
98
|
+
* continuous 3–10 second range, fractional seconds included, so it stays
|
|
99
|
+
* `number` — the adapter rejects out-of-range values at job creation,
|
|
100
|
+
* against the range entry below).
|
|
62
101
|
*
|
|
63
102
|
* @experimental Video generation is an experimental feature and may change.
|
|
64
103
|
*/
|
|
@@ -66,14 +105,20 @@ export type GeminiVideoModelDurationByName = {
|
|
|
66
105
|
'veo-3.1-generate-preview': 4 | 6 | 8;
|
|
67
106
|
'veo-3.1-fast-generate-preview': 4 | 6 | 8;
|
|
68
107
|
'veo-3.1-lite-generate-preview': 4 | 6 | 8;
|
|
108
|
+
'gemini-omni-flash-preview': number;
|
|
69
109
|
};
|
|
70
110
|
/**
|
|
71
111
|
* Runtime duration table backing `availableDurations()` / `snapDuration()`.
|
|
72
112
|
*
|
|
73
|
-
*
|
|
113
|
+
* Veo values are curated from the official docs
|
|
74
114
|
* (https://ai.google.dev/gemini-api/docs/video) — the Gemini OpenAPI spec
|
|
75
115
|
* types the `:predictLongRunning` request's `parameters` as unconstrained,
|
|
76
116
|
* so it carries no per-model duration information to derive these from.
|
|
117
|
+
* Omni Flash's 3–10s range was verified against the live API
|
|
118
|
+
* (2026-07-02): `response_format.duration` takes a `"<seconds>s"` string,
|
|
119
|
+
* fractional values are accepted, out-of-range values are rejected with
|
|
120
|
+
* "minimum allowed 3s" / "maximum allowed 10s", and omitting it defaults
|
|
121
|
+
* to a 10-second clip.
|
|
77
122
|
*
|
|
78
123
|
* @experimental Video generation is an experimental feature and may change.
|
|
79
124
|
*/
|
|
@@ -81,7 +126,7 @@ export declare const GEMINI_VIDEO_DURATIONS: {
|
|
|
81
126
|
readonly [TModel in GeminiVideoModel]: DurationOptions<GeminiVideoModelDurationByName[TModel]>;
|
|
82
127
|
};
|
|
83
128
|
/**
|
|
84
|
-
* Look up the duration options for a
|
|
129
|
+
* Look up the duration options for a Gemini video model.
|
|
85
130
|
*
|
|
86
131
|
* @experimental Video generation is an experimental feature and may change.
|
|
87
132
|
*/
|
|
@@ -1,13 +1,26 @@
|
|
|
1
|
+
import { GEMINI_INTERACTIONS_VIDEO_MODELS } from "../model-meta.js";
|
|
2
|
+
function isInteractionsVideoModel(model) {
|
|
3
|
+
return GEMINI_INTERACTIONS_VIDEO_MODELS.includes(
|
|
4
|
+
model
|
|
5
|
+
);
|
|
6
|
+
}
|
|
1
7
|
const GEMINI_VIDEO_DURATIONS = {
|
|
2
8
|
"veo-3.1-generate-preview": { kind: "discrete", values: [4, 6, 8] },
|
|
3
9
|
"veo-3.1-fast-generate-preview": { kind: "discrete", values: [4, 6, 8] },
|
|
4
|
-
"veo-3.1-lite-generate-preview": { kind: "discrete", values: [4, 6, 8] }
|
|
10
|
+
"veo-3.1-lite-generate-preview": { kind: "discrete", values: [4, 6, 8] },
|
|
11
|
+
"gemini-omni-flash-preview": {
|
|
12
|
+
kind: "range",
|
|
13
|
+
min: 3,
|
|
14
|
+
max: 10,
|
|
15
|
+
unit: "seconds"
|
|
16
|
+
}
|
|
5
17
|
};
|
|
6
18
|
function getGeminiVideoDurationOptions(model) {
|
|
7
19
|
return GEMINI_VIDEO_DURATIONS[model];
|
|
8
20
|
}
|
|
9
21
|
export {
|
|
10
22
|
GEMINI_VIDEO_DURATIONS,
|
|
11
|
-
getGeminiVideoDurationOptions
|
|
23
|
+
getGeminiVideoDurationOptions,
|
|
24
|
+
isInteractionsVideoModel
|
|
12
25
|
};
|
|
13
26
|
//# sourceMappingURL=video-provider-options.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"video-provider-options.js","sources":["../../../src/video/video-provider-options.ts"],"sourcesContent":["/**\n * Gemini
|
|
1
|
+
{"version":3,"file":"video-provider-options.js","sources":["../../../src/video/video-provider-options.ts"],"sourcesContent":["/**\n * Gemini Video Generation Provider Options\n *\n * Covers two request paths behind the one video adapter:\n * - Veo models — long-running operations via `:predictLongRunning`\n * (https://ai.google.dev/gemini-api/docs/video)\n * - Gemini Omni Flash — background jobs via the Interactions API\n * (https://ai.google.dev/gemini-api/docs/omni)\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nimport { GEMINI_INTERACTIONS_VIDEO_MODELS } from '../model-meta'\nimport type { DurationOptions } from '@tanstack/ai/adapters'\nimport type { GenerateVideosConfig, Interactions } from '@google/genai'\nimport type { GEMINI_VIDEO_MODELS } from '../model-meta'\n\n/**\n * Model type for Gemini video generation (Veo + Omni Flash).\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GeminiVideoModel = (typeof GEMINI_VIDEO_MODELS)[number]\n\n/**\n * Video models served by the Interactions API (Gemini Omni Flash) rather\n * than Veo's `:predictLongRunning` operations flow.\n * @experimental Omni video generation is an experimental feature and may change.\n */\nexport type GeminiInteractionsVideoModel =\n (typeof GEMINI_INTERACTIONS_VIDEO_MODELS)[number]\n\n/**\n * Runtime guard for the Interactions-served video models.\n * @experimental Omni video generation is an experimental feature and may change.\n */\nexport function isInteractionsVideoModel(\n model: GeminiVideoModel,\n): model is GeminiInteractionsVideoModel {\n return (GEMINI_INTERACTIONS_VIDEO_MODELS as ReadonlyArray<string>).includes(\n model,\n )\n}\n\n/**\n * Supported aspect ratios for Gemini video generation. This is the `size`\n * value for the Gemini video adapter — both Veo and Omni Flash express\n * output shape as an aspect ratio (plus an optional `resolution` in Veo's\n * `modelOptions`), not pixel dimensions.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GeminiVideoSize = '16:9' | '9:16'\n\n/**\n * Provider-specific options for Gemini Veo video generation.\n *\n * Derived from the SDK's `GenerateVideosConfig`, minus the fields the\n * adapter manages itself:\n * - `durationSeconds` — set via the typed top-level `duration` option\n * (use `adapter.snapDuration(seconds)` to coerce raw seconds)\n * - `aspectRatio` — set via the top-level `size` option\n * - `lastFrame` / `referenceImages` — set via image parts in the `prompt`\n * with `metadata.role: 'end_frame'` / `'reference'`\n * - `httpOptions` / `abortSignal` — client-level transport concerns\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GeminiVideoProviderOptions = Omit<\n GenerateVideosConfig,\n | 'durationSeconds'\n | 'aspectRatio'\n | 'lastFrame'\n | 'referenceImages'\n | 'httpOptions'\n | 'abortSignal'\n>\n\n/**\n * Provider-specific options for Gemini Omni Flash video generation on the\n * Interactions API.\n *\n * Derived from the SDK's `Interactions.CreateModelInteractionParamsNonStreaming`,\n * minus the fields the adapter manages itself:\n * - `model` / `input` — set from the adapter's model and the `prompt`\n * - `stream` / `background` — the adapter always creates a background job\n * and polls it through the `generateVideo` jobs API\n * - `response_modalities` / `response_format` — the adapter requests video\n * output and maps the top-level `size` option onto\n * `response_format.aspect_ratio`\n * - `tools` / `response_mime_type` — not applicable to video generation\n *\n * Notable passthroughs:\n * - `previous_interaction_id` — conversational video editing: chain a new\n * prompt onto a prior Omni interaction to refine its video\n * - `generation_config.video_config.task` — pin the task mode\n * (`'text_to_video' | 'image_to_video' | 'reference_to_video' | 'edit'`)\n * instead of letting the model infer it\n *\n * @experimental Omni video generation is an experimental feature and may change.\n */\nexport type GeminiOmniVideoProviderOptions = Omit<\n Interactions.CreateModelInteractionParamsNonStreaming,\n | 'model'\n | 'input'\n | 'stream'\n | 'background'\n | 'response_modalities'\n | 'response_format'\n | 'response_mime_type'\n | 'tools'\n>\n\n/**\n * Model-specific provider options mapping.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GeminiVideoModelProviderOptionsByName = {\n [TModel in GeminiVideoModel]: TModel extends GeminiInteractionsVideoModel\n ? GeminiOmniVideoProviderOptions\n : GeminiVideoProviderOptions\n}\n\n/**\n * Model-specific size (aspect ratio) mapping.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GeminiVideoModelSizeByName = {\n [TModel in GeminiVideoModel]: GeminiVideoSize\n}\n\n/**\n * Per-model prompt input modalities. Every Veo model accepts image\n * conditioning inputs (first frame, last frame, reference images) alongside\n * the text prompt. Omni Flash additionally accepts video inputs (short\n * reference clips / videos to edit).\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GeminiVideoModelInputModalitiesByName = {\n [TModel in GeminiVideoModel]: TModel extends GeminiInteractionsVideoModel\n ? readonly ['image', 'video']\n : readonly ['image']\n}\n\n/**\n * Per-model duration unions (seconds, as numbers — Veo's\n * `parameters.durationSeconds` field is numeric; Omni Flash accepts a\n * continuous 3–10 second range, fractional seconds included, so it stays\n * `number` — the adapter rejects out-of-range values at job creation,\n * against the range entry below).\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GeminiVideoModelDurationByName = {\n 'veo-3.1-generate-preview': 4 | 6 | 8\n 'veo-3.1-fast-generate-preview': 4 | 6 | 8\n 'veo-3.1-lite-generate-preview': 4 | 6 | 8\n 'gemini-omni-flash-preview': number\n}\n\n/**\n * Runtime duration table backing `availableDurations()` / `snapDuration()`.\n *\n * Veo values are curated from the official docs\n * (https://ai.google.dev/gemini-api/docs/video) — the Gemini OpenAPI spec\n * types the `:predictLongRunning` request's `parameters` as unconstrained,\n * so it carries no per-model duration information to derive these from.\n * Omni Flash's 3–10s range was verified against the live API\n * (2026-07-02): `response_format.duration` takes a `\"<seconds>s\"` string,\n * fractional values are accepted, out-of-range values are rejected with\n * \"minimum allowed 3s\" / \"maximum allowed 10s\", and omitting it defaults\n * to a 10-second clip.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport const GEMINI_VIDEO_DURATIONS: {\n readonly [TModel in GeminiVideoModel]: DurationOptions<\n GeminiVideoModelDurationByName[TModel]\n >\n} = {\n 'veo-3.1-generate-preview': { kind: 'discrete', values: [4, 6, 8] },\n 'veo-3.1-fast-generate-preview': { kind: 'discrete', values: [4, 6, 8] },\n 'veo-3.1-lite-generate-preview': { kind: 'discrete', values: [4, 6, 8] },\n 'gemini-omni-flash-preview': {\n kind: 'range',\n min: 3,\n max: 10,\n unit: 'seconds',\n },\n}\n\n/**\n * Look up the duration options for a Gemini video model.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport function getGeminiVideoDurationOptions<TModel extends GeminiVideoModel>(\n model: TModel,\n): DurationOptions<GeminiVideoModelDurationByName[TModel]> {\n return GEMINI_VIDEO_DURATIONS[model]\n}\n"],"names":[],"mappings":";AAkCO,SAAS,yBACd,OACuC;AACvC,SAAQ,iCAA2D;AAAA,IACjE;AAAA,EAAA;AAEJ;AAwIO,MAAM,yBAIT;AAAA,EACF,4BAA4B,EAAE,MAAM,YAAY,QAAQ,CAAC,GAAG,GAAG,CAAC,EAAA;AAAA,EAChE,iCAAiC,EAAE,MAAM,YAAY,QAAQ,CAAC,GAAG,GAAG,CAAC,EAAA;AAAA,EACrE,iCAAiC,EAAE,MAAM,YAAY,QAAQ,CAAC,GAAG,GAAG,CAAC,EAAA;AAAA,EACrE,6BAA6B;AAAA,IAC3B,MAAM;AAAA,IACN,KAAK;AAAA,IACL,KAAK;AAAA,IACL,MAAM;AAAA,EAAA;AAEV;AAOO,SAAS,8BACd,OACyD;AACzD,SAAO,uBAAuB,KAAK;AACrC;"}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tanstack/ai-gemini",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.20.1",
|
|
4
4
|
"description": "Google Gemini adapter for TanStack AI chat, images, speech, audio generation, and structured outputs.",
|
|
5
5
|
"author": "Tanner Linsley",
|
|
6
6
|
"license": "MIT",
|
|
@@ -54,18 +54,18 @@
|
|
|
54
54
|
"text-to-speech"
|
|
55
55
|
],
|
|
56
56
|
"dependencies": {
|
|
57
|
-
"@google/genai": "^2.
|
|
57
|
+
"@google/genai": "^2.10.0",
|
|
58
58
|
"partial-json": "^0.1.7",
|
|
59
59
|
"@tanstack/ai-utils": "0.3.1"
|
|
60
60
|
},
|
|
61
61
|
"peerDependencies": {
|
|
62
|
-
"@tanstack/ai": "^0.
|
|
62
|
+
"@tanstack/ai": "^0.42.0"
|
|
63
63
|
},
|
|
64
64
|
"devDependencies": {
|
|
65
65
|
"@vitest/coverage-v8": "4.0.14",
|
|
66
66
|
"vite": "^7.3.3",
|
|
67
67
|
"zod": "^4.2.0",
|
|
68
|
-
"@tanstack/ai": "0.
|
|
68
|
+
"@tanstack/ai": "0.42.0"
|
|
69
69
|
},
|
|
70
70
|
"scripts": {
|
|
71
71
|
"build": "vite build",
|
package/src/adapters/audio.ts
CHANGED
|
@@ -11,7 +11,7 @@ import type {
|
|
|
11
11
|
AudioGenerationResult,
|
|
12
12
|
} from '@tanstack/ai'
|
|
13
13
|
import type { GoogleGenAI } from '@google/genai'
|
|
14
|
-
import type { GeminiClientConfig } from '../utils'
|
|
14
|
+
import type { GeminiClientConfig } from '../utils/client'
|
|
15
15
|
|
|
16
16
|
/**
|
|
17
17
|
* Provider options for Gemini Lyria music generation.
|