@tanstack/ai-gemini 0.19.1 → 0.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/audio.d.ts +1 -1
- package/dist/esm/adapters/audio.js.map +1 -1
- package/dist/esm/adapters/image.d.ts +1 -1
- package/dist/esm/adapters/image.js +17 -39
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/adapters/summarize.d.ts +1 -1
- package/dist/esm/adapters/summarize.js.map +1 -1
- package/dist/esm/adapters/text.d.ts +1 -1
- package/dist/esm/adapters/text.js.map +1 -1
- package/dist/esm/adapters/tts.d.ts +1 -1
- package/dist/esm/adapters/tts.js.map +1 -1
- package/dist/esm/adapters/video.d.ts +60 -11
- package/dist/esm/adapters/video.js +205 -6
- package/dist/esm/adapters/video.js.map +1 -1
- package/dist/esm/experimental/text-interactions/adapter.d.ts +1 -1
- package/dist/esm/experimental/text-interactions/adapter.js.map +1 -1
- package/dist/esm/index.d.ts +6 -3
- package/dist/esm/index.js +9 -3
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/model-meta.d.ts +11 -3
- package/dist/esm/model-meta.js +9 -1
- package/dist/esm/model-meta.js.map +1 -1
- package/dist/esm/realtime/adapter.d.ts +22 -0
- package/dist/esm/realtime/adapter.js +233 -0
- package/dist/esm/realtime/adapter.js.map +1 -0
- package/dist/esm/realtime/client.d.ts +98 -0
- package/dist/esm/realtime/client.js +389 -0
- package/dist/esm/realtime/client.js.map +1 -0
- package/dist/esm/realtime/index.d.ts +3 -0
- package/dist/esm/realtime/token.d.ts +26 -0
- package/dist/esm/realtime/token.js +39 -0
- package/dist/esm/realtime/token.js.map +1 -0
- package/dist/esm/realtime/types.d.ts +51 -0
- package/dist/esm/realtime/utils.d.ts +40 -0
- package/dist/esm/realtime/utils.js +350 -0
- package/dist/esm/realtime/utils.js.map +1 -0
- package/dist/esm/video/video-provider-options.d.ts +59 -14
- package/dist/esm/video/video-provider-options.js +15 -2
- package/dist/esm/video/video-provider-options.js.map +1 -1
- package/package.json +4 -4
- package/src/adapters/audio.ts +1 -1
- package/src/adapters/image.ts +25 -49
- package/src/adapters/summarize.ts +1 -1
- package/src/adapters/text.ts +1 -1
- package/src/adapters/tts.ts +1 -1
- package/src/adapters/video.ts +333 -16
- package/src/experimental/text-interactions/adapter.ts +2 -2
- package/src/index.ts +20 -2
- package/src/model-meta.ts +45 -2
- package/src/realtime/adapter.ts +311 -0
- package/src/realtime/client.ts +547 -0
- package/src/realtime/index.ts +14 -0
- package/src/realtime/token.ts +70 -0
- package/src/realtime/types.ts +94 -0
- package/src/realtime/utils.ts +439 -0
- package/src/video/video-provider-options.ts +95 -15
|
@@ -0,0 +1,547 @@
|
|
|
1
|
+
import { convertSchemaToJsonSchema } from '@tanstack/ai'
|
|
2
|
+
import {
|
|
3
|
+
ActivityHandling,
|
|
4
|
+
EndSensitivity,
|
|
5
|
+
Modality,
|
|
6
|
+
StartSensitivity,
|
|
7
|
+
TurnCoverage,
|
|
8
|
+
} from '@google/genai'
|
|
9
|
+
import type {
|
|
10
|
+
ContextWindowCompressionConfig,
|
|
11
|
+
FunctionDeclaration,
|
|
12
|
+
FunctionResponse,
|
|
13
|
+
LiveClientMessage,
|
|
14
|
+
LiveServerGoAway,
|
|
15
|
+
LiveServerMessage,
|
|
16
|
+
LiveServerSessionResumptionUpdate,
|
|
17
|
+
LiveServerToolCall,
|
|
18
|
+
ThinkingConfig,
|
|
19
|
+
UsageMetadata,
|
|
20
|
+
} from '@google/genai'
|
|
21
|
+
import type {
|
|
22
|
+
AnyClientTool,
|
|
23
|
+
RealtimeSessionConfig,
|
|
24
|
+
RealtimeToken,
|
|
25
|
+
RealtimeToolConfig,
|
|
26
|
+
} from '@tanstack/ai'
|
|
27
|
+
import type {
|
|
28
|
+
GeminiRealtimeModel,
|
|
29
|
+
GeminiRealtimeProviderOptions,
|
|
30
|
+
GeminiRealtimeVoice,
|
|
31
|
+
} from './types'
|
|
32
|
+
|
|
33
|
+
/** Build a Gemini FunctionDeclaration from an isomorphic client tool (Zod). */
|
|
34
|
+
function clientToolToDeclaration(tool: AnyClientTool): FunctionDeclaration {
|
|
35
|
+
return {
|
|
36
|
+
name: tool.name,
|
|
37
|
+
description: tool.description,
|
|
38
|
+
parametersJsonSchema: convertSchemaToJsonSchema(tool.inputSchema),
|
|
39
|
+
responseJsonSchema: convertSchemaToJsonSchema(tool.outputSchema),
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/** Build a Gemini FunctionDeclaration from an already-serialized tool config. */
|
|
44
|
+
function toolConfigToDeclaration(
|
|
45
|
+
tool: RealtimeToolConfig,
|
|
46
|
+
): FunctionDeclaration {
|
|
47
|
+
return {
|
|
48
|
+
name: tool.name,
|
|
49
|
+
description: tool.description,
|
|
50
|
+
parametersJsonSchema: tool.inputSchema,
|
|
51
|
+
responseJsonSchema: tool.outputSchema,
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
interface LiveResponsePayloads {
|
|
56
|
+
text: string
|
|
57
|
+
thought: string
|
|
58
|
+
audio: { audioData: string; transcript: string }
|
|
59
|
+
setup_complete: string
|
|
60
|
+
interrupted: string
|
|
61
|
+
turn_complete: string
|
|
62
|
+
tool_call: LiveServerToolCall
|
|
63
|
+
session_resumption_update: LiveServerSessionResumptionUpdate
|
|
64
|
+
go_away: LiveServerGoAway
|
|
65
|
+
usage_metadata: UsageMetadata
|
|
66
|
+
error: string
|
|
67
|
+
input_transcription: { text: string; finished: boolean }
|
|
68
|
+
output_transcription: { text: string; finished: boolean }
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
export type MultimodalLiveResponseType = keyof LiveResponsePayloads
|
|
72
|
+
|
|
73
|
+
export type LiveResponse = {
|
|
74
|
+
[K in MultimodalLiveResponseType]: {
|
|
75
|
+
type: K
|
|
76
|
+
data: LiveResponsePayloads[K]
|
|
77
|
+
endOfTurn: boolean
|
|
78
|
+
}
|
|
79
|
+
}[MultimodalLiveResponseType]
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Parses response messages from the Gemini Live API
|
|
83
|
+
*/
|
|
84
|
+
/**
|
|
85
|
+
* Parses ALL response types from a single server message.
|
|
86
|
+
* The server can now bundle multiple fields (e.g. audio + transcription)
|
|
87
|
+
* in the same message. Returns an array of response objects.
|
|
88
|
+
*/
|
|
89
|
+
export function parseResponseMessages(
|
|
90
|
+
data: LiveServerMessage,
|
|
91
|
+
): Array<LiveResponse> {
|
|
92
|
+
const responses: Array<LiveResponse> = []
|
|
93
|
+
const serverContent = data.serverContent
|
|
94
|
+
const parts = serverContent?.modelTurn?.parts
|
|
95
|
+
|
|
96
|
+
// Setup complete (exclusive — no other fields expected)
|
|
97
|
+
if (data.setupComplete) {
|
|
98
|
+
responses.push({ type: 'setup_complete', data: '', endOfTurn: false })
|
|
99
|
+
return responses
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
// Tool call (exclusive)
|
|
103
|
+
if (data.toolCall) {
|
|
104
|
+
responses.push({
|
|
105
|
+
type: 'tool_call',
|
|
106
|
+
data: data.toolCall,
|
|
107
|
+
endOfTurn: false,
|
|
108
|
+
})
|
|
109
|
+
return responses
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
if (data.sessionResumptionUpdate) {
|
|
113
|
+
responses.push({
|
|
114
|
+
type: 'session_resumption_update',
|
|
115
|
+
data: data.sessionResumptionUpdate,
|
|
116
|
+
endOfTurn: false,
|
|
117
|
+
})
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
if (data.goAway) {
|
|
121
|
+
responses.push({ type: 'go_away', data: data.goAway, endOfTurn: false })
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
if (data.usageMetadata) {
|
|
125
|
+
responses.push({
|
|
126
|
+
type: 'usage_metadata',
|
|
127
|
+
data: data.usageMetadata,
|
|
128
|
+
endOfTurn: false,
|
|
129
|
+
})
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
// Audio data from model turn parts
|
|
133
|
+
if (parts?.length) {
|
|
134
|
+
for (const part of parts) {
|
|
135
|
+
if (part.inlineData?.data) {
|
|
136
|
+
responses.push({
|
|
137
|
+
type: 'audio',
|
|
138
|
+
data: {
|
|
139
|
+
audioData: part.inlineData.data,
|
|
140
|
+
// The transcription is independent to the model turn, which means it doesn't imply any ordering between transcription and model turn.
|
|
141
|
+
transcript: '',
|
|
142
|
+
},
|
|
143
|
+
endOfTurn: false,
|
|
144
|
+
})
|
|
145
|
+
} else if (part.text) {
|
|
146
|
+
responses.push({
|
|
147
|
+
type: part.thought ? 'thought' : 'text',
|
|
148
|
+
data: part.text,
|
|
149
|
+
endOfTurn: false,
|
|
150
|
+
})
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
// Transcriptions — checked independently, NOT in else-if with audio
|
|
156
|
+
if (serverContent?.inputTranscription) {
|
|
157
|
+
responses.push({
|
|
158
|
+
type: 'input_transcription',
|
|
159
|
+
data: {
|
|
160
|
+
text: serverContent.inputTranscription.text || '',
|
|
161
|
+
finished: serverContent.inputTranscription.finished || false,
|
|
162
|
+
},
|
|
163
|
+
endOfTurn: false,
|
|
164
|
+
})
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
if (serverContent?.outputTranscription) {
|
|
168
|
+
responses.push({
|
|
169
|
+
type: 'output_transcription',
|
|
170
|
+
data: {
|
|
171
|
+
text: serverContent.outputTranscription.text || '',
|
|
172
|
+
finished: serverContent.outputTranscription.finished || false,
|
|
173
|
+
},
|
|
174
|
+
endOfTurn: false,
|
|
175
|
+
})
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
// Interrupted
|
|
179
|
+
if (serverContent?.interrupted) {
|
|
180
|
+
responses.push({ type: 'interrupted', data: '', endOfTurn: false })
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
// Turn complete
|
|
184
|
+
if (serverContent?.turnComplete) {
|
|
185
|
+
responses.push({ type: 'turn_complete', data: '', endOfTurn: true })
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
return responses
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
export class GeminiLiveClient {
|
|
192
|
+
private token: string | null = null
|
|
193
|
+
private model: GeminiRealtimeModel | null = null
|
|
194
|
+
|
|
195
|
+
private readonly responseModalities: Array<Modality> = [Modality.AUDIO]
|
|
196
|
+
private systemInstructions = ''
|
|
197
|
+
private googleGrounding = false
|
|
198
|
+
private voiceName: GeminiRealtimeVoice = 'Puck'
|
|
199
|
+
private temperature = 1.0
|
|
200
|
+
private inputAudioTranscription = false
|
|
201
|
+
private outputAudioTranscription = false
|
|
202
|
+
private contextWindowCompression: ContextWindowCompressionConfig | undefined
|
|
203
|
+
private proactiveAudio = false
|
|
204
|
+
private enableAffectiveDialog = false
|
|
205
|
+
private thinkingConfig: ThinkingConfig | undefined
|
|
206
|
+
private speechLanguageCode: string | undefined
|
|
207
|
+
|
|
208
|
+
private maxOutputTokens: number | undefined
|
|
209
|
+
private functionDeclarations: Array<FunctionDeclaration> = []
|
|
210
|
+
|
|
211
|
+
private readonly automaticActivityDetection = {
|
|
212
|
+
disabled: false,
|
|
213
|
+
silence_duration_ms: 2000,
|
|
214
|
+
prefix_padding_ms: 500,
|
|
215
|
+
end_of_speech_sensitivity: EndSensitivity.END_SENSITIVITY_UNSPECIFIED,
|
|
216
|
+
start_of_speech_sensitivity: StartSensitivity.START_SENSITIVITY_UNSPECIFIED,
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
private readonly activityHandling =
|
|
220
|
+
ActivityHandling.ACTIVITY_HANDLING_UNSPECIFIED
|
|
221
|
+
|
|
222
|
+
private webSocket: WebSocket | null = null
|
|
223
|
+
private lastResumptionUpdate: LiveServerSessionResumptionUpdate | null = null
|
|
224
|
+
private setupComplete = false
|
|
225
|
+
private connected = false
|
|
226
|
+
|
|
227
|
+
public onReceiveResponse: (response: LiveResponse) => void = () => {}
|
|
228
|
+
public onOpen: () => void = () => {}
|
|
229
|
+
public onClose: () => void = () => {}
|
|
230
|
+
public onError: (error: Error) => void = () => {}
|
|
231
|
+
|
|
232
|
+
constructor(
|
|
233
|
+
token: string,
|
|
234
|
+
model: GeminiRealtimeModel,
|
|
235
|
+
tools?: ReadonlyArray<AnyClientTool>,
|
|
236
|
+
) {
|
|
237
|
+
this.token = token
|
|
238
|
+
this.model = model
|
|
239
|
+
|
|
240
|
+
if (tools) {
|
|
241
|
+
this.functionDeclarations = tools.map(clientToolToDeclaration)
|
|
242
|
+
}
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
get isConnected() {
|
|
246
|
+
return this.connected
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
get isSetupComplete() {
|
|
250
|
+
return this.setupComplete
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
/**
|
|
254
|
+
* Connection management
|
|
255
|
+
*/
|
|
256
|
+
connect(): Promise<void> {
|
|
257
|
+
return new Promise((resolve, reject) => {
|
|
258
|
+
const socket = new WebSocket(
|
|
259
|
+
`wss://generativelanguage.googleapis.com/ws/google.ai.generativelanguage.v1alpha.GenerativeService.BidiGenerateContentConstrained?access_token=${this.token}`,
|
|
260
|
+
)
|
|
261
|
+
this.webSocket = socket
|
|
262
|
+
|
|
263
|
+
// The browser fires `onerror` then `onclose` on an abnormal close; this
|
|
264
|
+
// flag stops the close handler from overwriting the surfaced error with a
|
|
265
|
+
// benign "closed" signal.
|
|
266
|
+
let errored = false
|
|
267
|
+
|
|
268
|
+
socket.onclose = () => {
|
|
269
|
+
this.connected = false
|
|
270
|
+
this.setupComplete = false
|
|
271
|
+
if (!errored) this.onClose()
|
|
272
|
+
reject(new Error('WebSocket closed before setup completed'))
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
socket.onerror = () => {
|
|
276
|
+
errored = true
|
|
277
|
+
this.connected = false
|
|
278
|
+
this.setupComplete = false
|
|
279
|
+
const error = new Error('Gemini realtime WebSocket connection error')
|
|
280
|
+
this.onError(error)
|
|
281
|
+
reject(error)
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
socket.onopen = () => {
|
|
285
|
+
this.connected = true
|
|
286
|
+
this.onOpen()
|
|
287
|
+
resolve()
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
socket.onmessage = (event) => {
|
|
291
|
+
void this.onReceiveMessage(event)
|
|
292
|
+
}
|
|
293
|
+
})
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
disconnect() {
|
|
297
|
+
if (this.webSocket) {
|
|
298
|
+
// Detach handlers first so a deliberate teardown (e.g. during a
|
|
299
|
+
// reconnect) doesn't emit a spurious close/error to the client.
|
|
300
|
+
this.webSocket.onclose = null
|
|
301
|
+
this.webSocket.onerror = null
|
|
302
|
+
this.webSocket.onopen = null
|
|
303
|
+
this.webSocket.onmessage = null
|
|
304
|
+
this.webSocket.close()
|
|
305
|
+
this.webSocket = null
|
|
306
|
+
}
|
|
307
|
+
this.connected = false
|
|
308
|
+
this.setupComplete = false
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
/**
|
|
312
|
+
* Session management
|
|
313
|
+
*/
|
|
314
|
+
sendInitialSetupMessage(resume = false) {
|
|
315
|
+
const tools = this.functionDeclarations
|
|
316
|
+
|
|
317
|
+
const setup: NonNullable<LiveClientMessage['setup']> = {
|
|
318
|
+
model: `models/${this.model}`,
|
|
319
|
+
generationConfig: {
|
|
320
|
+
responseModalities: this.responseModalities,
|
|
321
|
+
temperature: this.temperature,
|
|
322
|
+
speechConfig: {
|
|
323
|
+
languageCode: this.speechLanguageCode,
|
|
324
|
+
voiceConfig: {
|
|
325
|
+
prebuiltVoiceConfig: {
|
|
326
|
+
voiceName: this.voiceName,
|
|
327
|
+
},
|
|
328
|
+
},
|
|
329
|
+
},
|
|
330
|
+
enableAffectiveDialog: this.enableAffectiveDialog,
|
|
331
|
+
maxOutputTokens: this.maxOutputTokens,
|
|
332
|
+
thinkingConfig: this.thinkingConfig,
|
|
333
|
+
},
|
|
334
|
+
sessionResumption: {
|
|
335
|
+
transparent: true,
|
|
336
|
+
handle: resume ? this.lastResumptionUpdate?.newHandle : undefined,
|
|
337
|
+
},
|
|
338
|
+
contextWindowCompression: this.contextWindowCompression,
|
|
339
|
+
proactivity: {
|
|
340
|
+
proactiveAudio: this.proactiveAudio,
|
|
341
|
+
},
|
|
342
|
+
systemInstruction: { parts: [{ text: this.systemInstructions }] },
|
|
343
|
+
tools: [{ functionDeclarations: tools }],
|
|
344
|
+
realtimeInputConfig: {
|
|
345
|
+
automaticActivityDetection: {
|
|
346
|
+
disabled: this.automaticActivityDetection.disabled,
|
|
347
|
+
silenceDurationMs:
|
|
348
|
+
this.automaticActivityDetection.silence_duration_ms,
|
|
349
|
+
prefixPaddingMs: this.automaticActivityDetection.prefix_padding_ms,
|
|
350
|
+
endOfSpeechSensitivity:
|
|
351
|
+
this.automaticActivityDetection.end_of_speech_sensitivity,
|
|
352
|
+
startOfSpeechSensitivity:
|
|
353
|
+
this.automaticActivityDetection.start_of_speech_sensitivity,
|
|
354
|
+
},
|
|
355
|
+
activityHandling: this.activityHandling,
|
|
356
|
+
turnCoverage: TurnCoverage.TURN_INCLUDES_ONLY_ACTIVITY,
|
|
357
|
+
},
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
if (this.inputAudioTranscription) {
|
|
361
|
+
setup.inputAudioTranscription = {}
|
|
362
|
+
}
|
|
363
|
+
|
|
364
|
+
if (this.outputAudioTranscription) {
|
|
365
|
+
setup.outputAudioTranscription = {}
|
|
366
|
+
}
|
|
367
|
+
|
|
368
|
+
if (this.googleGrounding) {
|
|
369
|
+
// Currently can't have both Google Search with custom tools.
|
|
370
|
+
console.warn(
|
|
371
|
+
'Google Grounding enabled, removing custom function calls if any.',
|
|
372
|
+
)
|
|
373
|
+
setup.tools = [{ googleSearch: {} }]
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
this.sendMessage({ setup })
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
async restartSession(resume = false) {
|
|
380
|
+
this.disconnect()
|
|
381
|
+
await this.connect()
|
|
382
|
+
this.sendInitialSetupMessage(resume)
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
updateToken(token: RealtimeToken) {
|
|
386
|
+
this.token = token.token
|
|
387
|
+
|
|
388
|
+
// Restart completely with the new model, or resume the existing session.
|
|
389
|
+
const resume = !(token.config.model && this.model != token.config.model)
|
|
390
|
+
if (!resume) {
|
|
391
|
+
this.model = token.config.model as GeminiRealtimeModel
|
|
392
|
+
}
|
|
393
|
+
this.restartSession(resume).catch((err) =>
|
|
394
|
+
this.onError(err instanceof Error ? err : new Error(String(err))),
|
|
395
|
+
)
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
async updateSession(config: Partial<RealtimeSessionConfig>) {
|
|
399
|
+
// model can only be set during initial setup
|
|
400
|
+
if (config.model && !this.setupComplete) {
|
|
401
|
+
this.model = config.model as GeminiRealtimeModel
|
|
402
|
+
}
|
|
403
|
+
|
|
404
|
+
if (config.instructions) {
|
|
405
|
+
this.systemInstructions = config.instructions
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
if (config.tools) {
|
|
409
|
+
this.functionDeclarations = config.tools.map(toolConfigToDeclaration)
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
if (config.maxOutputTokens) {
|
|
413
|
+
// Gemini has no "inf" sentinel; treat it as "no explicit limit".
|
|
414
|
+
this.maxOutputTokens =
|
|
415
|
+
typeof config.maxOutputTokens === 'number'
|
|
416
|
+
? config.maxOutputTokens
|
|
417
|
+
: undefined
|
|
418
|
+
}
|
|
419
|
+
|
|
420
|
+
if (config.temperature) {
|
|
421
|
+
this.temperature = config.temperature
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
if (config.voice) {
|
|
425
|
+
this.voiceName = config.voice as GeminiRealtimeVoice
|
|
426
|
+
}
|
|
427
|
+
|
|
428
|
+
const providerOptions = config.providerOptions as
|
|
429
|
+
| GeminiRealtimeProviderOptions
|
|
430
|
+
| undefined
|
|
431
|
+
|
|
432
|
+
if (providerOptions?.googleGrounding) {
|
|
433
|
+
this.googleGrounding = providerOptions.googleGrounding
|
|
434
|
+
}
|
|
435
|
+
|
|
436
|
+
if (providerOptions?.proactiveAudio) {
|
|
437
|
+
this.proactiveAudio = providerOptions.proactiveAudio
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
if (providerOptions?.enableAffectiveDialog) {
|
|
441
|
+
this.enableAffectiveDialog = providerOptions.enableAffectiveDialog
|
|
442
|
+
}
|
|
443
|
+
|
|
444
|
+
if (providerOptions?.contextWindowCompression) {
|
|
445
|
+
this.contextWindowCompression = providerOptions.contextWindowCompression
|
|
446
|
+
}
|
|
447
|
+
|
|
448
|
+
if (providerOptions?.thinkingConfig) {
|
|
449
|
+
this.thinkingConfig = providerOptions.thinkingConfig
|
|
450
|
+
}
|
|
451
|
+
|
|
452
|
+
if (providerOptions?.languageCode) {
|
|
453
|
+
this.speechLanguageCode = providerOptions.languageCode
|
|
454
|
+
}
|
|
455
|
+
|
|
456
|
+
const includeTranscription =
|
|
457
|
+
config.outputModalities?.includes('text') || false
|
|
458
|
+
this.inputAudioTranscription = includeTranscription
|
|
459
|
+
this.outputAudioTranscription = includeTranscription
|
|
460
|
+
|
|
461
|
+
if (!this.setupComplete) {
|
|
462
|
+
this.sendInitialSetupMessage()
|
|
463
|
+
} else {
|
|
464
|
+
return this.restartSession(true)
|
|
465
|
+
}
|
|
466
|
+
}
|
|
467
|
+
|
|
468
|
+
/**
|
|
469
|
+
* Message transmission & receiving
|
|
470
|
+
*/
|
|
471
|
+
sendMessage(message: LiveClientMessage) {
|
|
472
|
+
if (this.webSocket?.readyState === WebSocket.OPEN) {
|
|
473
|
+
this.webSocket.send(JSON.stringify(message))
|
|
474
|
+
} else {
|
|
475
|
+
this.onError(
|
|
476
|
+
new Error('Cannot send message: Gemini realtime socket is not open'),
|
|
477
|
+
)
|
|
478
|
+
}
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
async onReceiveMessage(messageEvent: MessageEvent) {
|
|
482
|
+
let jsonData
|
|
483
|
+
if (messageEvent.data instanceof Blob) {
|
|
484
|
+
jsonData = await messageEvent.data.text()
|
|
485
|
+
} else if (messageEvent.data instanceof ArrayBuffer) {
|
|
486
|
+
jsonData = new TextDecoder().decode(messageEvent.data)
|
|
487
|
+
} else {
|
|
488
|
+
jsonData = messageEvent.data
|
|
489
|
+
}
|
|
490
|
+
|
|
491
|
+
try {
|
|
492
|
+
const messageData = JSON.parse(jsonData)
|
|
493
|
+
// Parse all response types from this message (audio + transcription can coexist)
|
|
494
|
+
const responses = parseResponseMessages(messageData)
|
|
495
|
+
for (const response of responses) {
|
|
496
|
+
if (
|
|
497
|
+
response.type === 'session_resumption_update' &&
|
|
498
|
+
response.data.resumable
|
|
499
|
+
) {
|
|
500
|
+
this.lastResumptionUpdate = response.data
|
|
501
|
+
}
|
|
502
|
+
if (response.type === 'setup_complete') {
|
|
503
|
+
this.setupComplete = true
|
|
504
|
+
}
|
|
505
|
+
this.onReceiveResponse(response)
|
|
506
|
+
}
|
|
507
|
+
} catch (err) {
|
|
508
|
+
this.onError(err instanceof Error ? err : new Error(String(err)))
|
|
509
|
+
}
|
|
510
|
+
}
|
|
511
|
+
|
|
512
|
+
sendRealtimeInputMessage(data: string, mimeType: string) {
|
|
513
|
+
const blob = { mimeType, data }
|
|
514
|
+
|
|
515
|
+
if (mimeType.startsWith('audio/')) {
|
|
516
|
+
this.sendMessage({ realtimeInput: { audio: blob } })
|
|
517
|
+
} else if (mimeType.startsWith('image/') || mimeType.startsWith('video/')) {
|
|
518
|
+
this.sendMessage({ realtimeInput: { video: blob } })
|
|
519
|
+
}
|
|
520
|
+
}
|
|
521
|
+
|
|
522
|
+
sendAudioMessage(base64PCM: string) {
|
|
523
|
+
this.sendRealtimeInputMessage(base64PCM, 'audio/pcm')
|
|
524
|
+
}
|
|
525
|
+
|
|
526
|
+
sendImageMessage(base64: string, mimeType = 'image/jpeg') {
|
|
527
|
+
this.sendRealtimeInputMessage(base64, mimeType)
|
|
528
|
+
}
|
|
529
|
+
|
|
530
|
+
sendTextMessage(text: string) {
|
|
531
|
+
const message: LiveClientMessage = {
|
|
532
|
+
realtimeInput: {
|
|
533
|
+
text,
|
|
534
|
+
},
|
|
535
|
+
}
|
|
536
|
+
this.sendMessage(message)
|
|
537
|
+
}
|
|
538
|
+
|
|
539
|
+
sendToolResponse(functionResponses: Array<FunctionResponse>) {
|
|
540
|
+
const message: LiveClientMessage = {
|
|
541
|
+
toolResponse: {
|
|
542
|
+
functionResponses,
|
|
543
|
+
},
|
|
544
|
+
}
|
|
545
|
+
this.sendMessage(message)
|
|
546
|
+
}
|
|
547
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
// Token adapter for server-side use
|
|
2
|
+
export { geminiRealtimeToken } from './token'
|
|
3
|
+
|
|
4
|
+
// Client adapter for browser use
|
|
5
|
+
export { geminiRealtime } from './adapter'
|
|
6
|
+
|
|
7
|
+
// Types
|
|
8
|
+
export type {
|
|
9
|
+
GeminiRealtimeModel,
|
|
10
|
+
GeminiRealtimeVoice,
|
|
11
|
+
GeminiRealtimeTokenOptions,
|
|
12
|
+
GeminiRealtimeOptions,
|
|
13
|
+
GeminiRealtimeProviderOptions,
|
|
14
|
+
} from './types'
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import { GoogleGenAI } from '@google/genai'
|
|
2
|
+
import { getGeminiApiKeyFromEnv } from '../utils'
|
|
3
|
+
import type { RealtimeToken, RealtimeTokenAdapter } from '@tanstack/ai'
|
|
4
|
+
import type { GeminiRealtimeTokenOptions } from './types'
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Creates a Google Gemini realtime token adapter.
|
|
8
|
+
*
|
|
9
|
+
* This adapter generates ephemeral tokens for client-side WebSocket connections.
|
|
10
|
+
*
|
|
11
|
+
* @param options - Configuration options for the realtime session
|
|
12
|
+
* @returns A RealtimeTokenAdapter for use with realtimeToken()
|
|
13
|
+
*
|
|
14
|
+
* @example
|
|
15
|
+
* ```typescript
|
|
16
|
+
* import { realtimeToken } from '@tanstack/ai'
|
|
17
|
+
* import { geminiRealtimeToken } from '@tanstack/ai-gemini'
|
|
18
|
+
*
|
|
19
|
+
* const token = await realtimeToken({
|
|
20
|
+
* adapter: geminiRealtimeToken({
|
|
21
|
+
* // Optional: constraint model config by token
|
|
22
|
+
* liveConnectConstraints: {
|
|
23
|
+
* model: 'gemini-3.1-flash-live-preview',
|
|
24
|
+
* },
|
|
25
|
+
* }),
|
|
26
|
+
* })
|
|
27
|
+
* ```
|
|
28
|
+
*/
|
|
29
|
+
export function geminiRealtimeToken(
|
|
30
|
+
options: GeminiRealtimeTokenOptions = {},
|
|
31
|
+
): RealtimeTokenAdapter {
|
|
32
|
+
const apiKey = getGeminiApiKeyFromEnv()
|
|
33
|
+
|
|
34
|
+
const client = new GoogleGenAI({
|
|
35
|
+
apiKey,
|
|
36
|
+
})
|
|
37
|
+
|
|
38
|
+
return {
|
|
39
|
+
provider: 'gemini',
|
|
40
|
+
async generateToken(): Promise<RealtimeToken> {
|
|
41
|
+
// Computed per call so a reused adapter doesn't mint tokens with a
|
|
42
|
+
// fixed (and eventually past) expiry. Defaults to 30 minutes.
|
|
43
|
+
const expireTime = options.expiresAt ?? Date.now() + 30 * 60 * 1000
|
|
44
|
+
|
|
45
|
+
const token = await client.authTokens.create({
|
|
46
|
+
config: {
|
|
47
|
+
uses: options.uses ?? 1,
|
|
48
|
+
expireTime: new Date(expireTime).toISOString(),
|
|
49
|
+
liveConnectConstraints: options.liveConnectConstraints,
|
|
50
|
+
httpOptions: {
|
|
51
|
+
apiVersion: 'v1alpha',
|
|
52
|
+
},
|
|
53
|
+
},
|
|
54
|
+
})
|
|
55
|
+
|
|
56
|
+
if (!token.name) {
|
|
57
|
+
throw new Error('Gemini realtime token creation failed')
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
return {
|
|
61
|
+
provider: 'gemini',
|
|
62
|
+
token: token.name,
|
|
63
|
+
expiresAt: expireTime,
|
|
64
|
+
config: {
|
|
65
|
+
model: options.liveConnectConstraints?.model,
|
|
66
|
+
},
|
|
67
|
+
}
|
|
68
|
+
},
|
|
69
|
+
}
|
|
70
|
+
}
|