@mastra/voice-google-gemini-live 0.14.6-alpha.4 → 0.14.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/dist/docs/SKILL.md +3 -3
- package/dist/docs/assets/SOURCE_MAP.json +1 -1
- package/dist/docs/references/guides-voice-overview.md +55 -106
- package/dist/docs/references/{reference-voice-google-gemini-live.md → integrations-voice-google.md} +305 -30
- package/package.json +6 -6
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,17 @@
|
|
|
1
1
|
# @mastra/voice-google-gemini-live
|
|
2
2
|
|
|
3
|
+
## 0.14.6
|
|
4
|
+
|
|
5
|
+
### Patch Changes
|
|
6
|
+
|
|
7
|
+
- dependencies updates: ([#20409](https://github.com/mastra-ai/mastra/pull/20409))
|
|
8
|
+
- Updated dependency [`google-auth-library@^10.9.1` ↗︎](https://www.npmjs.com/package/google-auth-library/v/10.9.1) (from `^10.9.0`, in `dependencies`)
|
|
9
|
+
|
|
10
|
+
- Fixed Gemini Live session readiness waits to release timeout resources after settling. ([#21083](https://github.com/mastra-ai/mastra/pull/21083))
|
|
11
|
+
|
|
12
|
+
- Updated dependencies [[`6bff877`](https://github.com/mastra-ai/mastra/commit/6bff877e214695ff8d9c84b06c13a6e6bcf9f1ed), [`f5a17d9`](https://github.com/mastra-ai/mastra/commit/f5a17d95c19e7d4149996932bd8d1905089f031d), [`5dba2a4`](https://github.com/mastra-ai/mastra/commit/5dba2a41600385751f5aace79878904e1972609d), [`9be8878`](https://github.com/mastra-ai/mastra/commit/9be8878dcf0388e84fc4873e0eec27bd49b881a4)]:
|
|
13
|
+
- @mastra/schema-compat@1.3.6
|
|
14
|
+
|
|
3
15
|
## 0.14.6-alpha.4
|
|
4
16
|
|
|
5
17
|
### Patch Changes
|
package/dist/docs/SKILL.md
CHANGED
|
@@ -3,7 +3,7 @@ name: mastra-voice-google-gemini-live
|
|
|
3
3
|
description: Documentation for @mastra/voice-google-gemini-live. Use when working with @mastra/voice-google-gemini-live APIs, configuration, or implementation.
|
|
4
4
|
metadata:
|
|
5
5
|
package: "@mastra/voice-google-gemini-live"
|
|
6
|
-
version: "0.14.6
|
|
6
|
+
version: "0.14.6"
|
|
7
7
|
---
|
|
8
8
|
|
|
9
9
|
## When to use
|
|
@@ -19,9 +19,9 @@ Read the individual reference documents for detailed explanations and code examp
|
|
|
19
19
|
- [Voice in Mastra](references/guides-voice-overview.md) - Overview of voice capabilities in Mastra, including text-to-speech, speech-to-text, and real-time speech-to-speech interactions.
|
|
20
20
|
- [Speech-to-Speech capabilities in Mastra](references/guides-voice-speech-to-speech.md) - Overview of speech-to-speech capabilities in Mastra, including real-time interactions and event-driven architecture.
|
|
21
21
|
|
|
22
|
-
###
|
|
22
|
+
### Integrations
|
|
23
23
|
|
|
24
|
-
- [
|
|
24
|
+
- [Google](references/integrations-voice-google.md) - Documentation for the Google Voice implementation, providing text-to-speech and speech-to-text capabilities with support for both API key and Vertex AI authentication.
|
|
25
25
|
|
|
26
26
|
|
|
27
27
|
Read [assets/SOURCE_MAP.json](assets/SOURCE_MAP.json) for source code references.
|
|
@@ -57,7 +57,7 @@ const audioStream = await voiceAgent.voice.speak(text, {
|
|
|
57
57
|
playAudio(audioStream)
|
|
58
58
|
```
|
|
59
59
|
|
|
60
|
-
Visit the [OpenAI Voice Reference](https://mastra.ai/
|
|
60
|
+
Visit the [OpenAI Voice Reference](https://mastra.ai/integrations/voice/openai) for more information on the OpenAI voice provider.
|
|
61
61
|
|
|
62
62
|
**Azure**:
|
|
63
63
|
|
|
@@ -84,7 +84,7 @@ const audioStream = await voiceAgent.voice.speak(text, {
|
|
|
84
84
|
playAudio(audioStream)
|
|
85
85
|
```
|
|
86
86
|
|
|
87
|
-
Visit the [Azure Voice Reference](https://mastra.ai/
|
|
87
|
+
Visit the [Azure Voice Reference](https://mastra.ai/integrations/voice/azure) for more information on the Azure voice provider.
|
|
88
88
|
|
|
89
89
|
**ElevenLabs**:
|
|
90
90
|
|
|
@@ -111,34 +111,7 @@ const audioStream = await voiceAgent.voice.speak(text, {
|
|
|
111
111
|
playAudio(audioStream)
|
|
112
112
|
```
|
|
113
113
|
|
|
114
|
-
Visit the [ElevenLabs Voice Reference](https://mastra.ai/
|
|
115
|
-
|
|
116
|
-
**PlayAI**:
|
|
117
|
-
|
|
118
|
-
```typescript
|
|
119
|
-
import { Agent } from '@mastra/core/agent'
|
|
120
|
-
import { PlayAIVoice } from '@mastra/voice-playai'
|
|
121
|
-
import { playAudio } from '@mastra/node-audio'
|
|
122
|
-
|
|
123
|
-
const voiceAgent = new Agent({
|
|
124
|
-
id: 'voice-agent',
|
|
125
|
-
name: 'Voice Agent',
|
|
126
|
-
instructions: 'You are a voice assistant that can help users with their tasks.',
|
|
127
|
-
model: 'openai/gpt-5.6-sol',
|
|
128
|
-
voice: new PlayAIVoice(),
|
|
129
|
-
})
|
|
130
|
-
|
|
131
|
-
const { text } = await voiceAgent.generate('What color is the sky?')
|
|
132
|
-
|
|
133
|
-
// Convert text to speech to an Audio Stream
|
|
134
|
-
const audioStream = await voiceAgent.voice.speak(text, {
|
|
135
|
-
speaker: 'default', // Optional: specify a speaker
|
|
136
|
-
})
|
|
137
|
-
|
|
138
|
-
playAudio(audioStream)
|
|
139
|
-
```
|
|
140
|
-
|
|
141
|
-
Visit the [PlayAI Voice Reference](https://mastra.ai/reference/voice/playai) for more information on the PlayAI voice provider.
|
|
114
|
+
Visit the [ElevenLabs Voice Reference](https://mastra.ai/integrations/voice/elevenlabs) for more information on the ElevenLabs voice provider.
|
|
142
115
|
|
|
143
116
|
**Google**:
|
|
144
117
|
|
|
@@ -165,7 +138,7 @@ const audioStream = await voiceAgent.voice.speak(text, {
|
|
|
165
138
|
playAudio(audioStream)
|
|
166
139
|
```
|
|
167
140
|
|
|
168
|
-
Visit the [Google Voice Reference](https://mastra.ai/
|
|
141
|
+
Visit the [Google Voice Reference](https://mastra.ai/integrations/voice/google) for more information on the Google voice provider.
|
|
169
142
|
|
|
170
143
|
**Cloudflare**:
|
|
171
144
|
|
|
@@ -192,7 +165,7 @@ const audioStream = await voiceAgent.voice.speak(text, {
|
|
|
192
165
|
playAudio(audioStream)
|
|
193
166
|
```
|
|
194
167
|
|
|
195
|
-
Visit the [Cloudflare Voice Reference](https://mastra.ai/
|
|
168
|
+
Visit the [Cloudflare Voice Reference](https://mastra.ai/integrations/voice/cloudflare) for more information on the Cloudflare voice provider.
|
|
196
169
|
|
|
197
170
|
**Deepgram**:
|
|
198
171
|
|
|
@@ -219,7 +192,7 @@ const audioStream = await voiceAgent.voice.speak(text, {
|
|
|
219
192
|
playAudio(audioStream)
|
|
220
193
|
```
|
|
221
194
|
|
|
222
|
-
Visit the [Deepgram Voice Reference](https://mastra.ai/
|
|
195
|
+
Visit the [Deepgram Voice Reference](https://mastra.ai/integrations/voice/deepgram) for more information on the Deepgram voice provider.
|
|
223
196
|
|
|
224
197
|
**Inworld**:
|
|
225
198
|
|
|
@@ -246,7 +219,7 @@ const audioStream = await voiceAgent.voice.speak(text, {
|
|
|
246
219
|
playAudio(audioStream)
|
|
247
220
|
```
|
|
248
221
|
|
|
249
|
-
Visit the [Inworld Voice Reference](https://mastra.ai/
|
|
222
|
+
Visit the [Inworld Voice Reference](https://mastra.ai/integrations/voice/inworld) for more information on the Inworld voice provider.
|
|
250
223
|
|
|
251
224
|
**Speechify**:
|
|
252
225
|
|
|
@@ -273,7 +246,7 @@ const audioStream = await voiceAgent.voice.speak(text, {
|
|
|
273
246
|
playAudio(audioStream)
|
|
274
247
|
```
|
|
275
248
|
|
|
276
|
-
Visit the [Speechify Voice Reference](https://mastra.ai/
|
|
249
|
+
Visit the [Speechify Voice Reference](https://mastra.ai/integrations/voice/speechify) for more information on the Speechify voice provider.
|
|
277
250
|
|
|
278
251
|
**Sarvam**:
|
|
279
252
|
|
|
@@ -300,7 +273,7 @@ const audioStream = await voiceAgent.voice.speak(text, {
|
|
|
300
273
|
playAudio(audioStream)
|
|
301
274
|
```
|
|
302
275
|
|
|
303
|
-
Visit the [Sarvam Voice Reference](https://mastra.ai/
|
|
276
|
+
Visit the [Sarvam Voice Reference](https://mastra.ai/integrations/voice/sarvam) for more information on the Sarvam voice provider.
|
|
304
277
|
|
|
305
278
|
**Murf**:
|
|
306
279
|
|
|
@@ -327,7 +300,7 @@ const audioStream = await voiceAgent.voice.speak(text, {
|
|
|
327
300
|
playAudio(audioStream)
|
|
328
301
|
```
|
|
329
302
|
|
|
330
|
-
Visit the [Murf Voice Reference](https://mastra.ai/
|
|
303
|
+
Visit the [Murf Voice Reference](https://mastra.ai/integrations/voice/murf) for more information on the Murf voice provider.
|
|
331
304
|
|
|
332
305
|
### Speech to Text (STT)
|
|
333
306
|
|
|
@@ -363,7 +336,7 @@ console.log(`User said: ${transcript}`)
|
|
|
363
336
|
const { text } = await voiceAgent.generate(transcript)
|
|
364
337
|
```
|
|
365
338
|
|
|
366
|
-
Visit the [OpenAI Voice Reference](https://mastra.ai/
|
|
339
|
+
Visit the [OpenAI Voice Reference](https://mastra.ai/integrations/voice/openai) for more information on the OpenAI voice provider.
|
|
367
340
|
|
|
368
341
|
**Azure**:
|
|
369
342
|
|
|
@@ -392,7 +365,7 @@ console.log(`User said: ${transcript}`)
|
|
|
392
365
|
const { text } = await voiceAgent.generate(transcript)
|
|
393
366
|
```
|
|
394
367
|
|
|
395
|
-
Visit the [Azure Voice Reference](https://mastra.ai/
|
|
368
|
+
Visit the [Azure Voice Reference](https://mastra.ai/integrations/voice/azure) for more information on the Azure voice provider.
|
|
396
369
|
|
|
397
370
|
**ElevenLabs**:
|
|
398
371
|
|
|
@@ -420,7 +393,7 @@ console.log(`User said: ${transcript}`)
|
|
|
420
393
|
const { text } = await voiceAgent.generate(transcript)
|
|
421
394
|
```
|
|
422
395
|
|
|
423
|
-
Visit the [ElevenLabs Voice Reference](https://mastra.ai/
|
|
396
|
+
Visit the [ElevenLabs Voice Reference](https://mastra.ai/integrations/voice/elevenlabs) for more information on the ElevenLabs voice provider.
|
|
424
397
|
|
|
425
398
|
**Google**:
|
|
426
399
|
|
|
@@ -448,7 +421,7 @@ console.log(`User said: ${transcript}`)
|
|
|
448
421
|
const { text } = await voiceAgent.generate(transcript)
|
|
449
422
|
```
|
|
450
423
|
|
|
451
|
-
Visit the [Google Voice Reference](https://mastra.ai/
|
|
424
|
+
Visit the [Google Voice Reference](https://mastra.ai/integrations/voice/google) for more information on the Google voice provider.
|
|
452
425
|
|
|
453
426
|
**Cloudflare**:
|
|
454
427
|
|
|
@@ -476,7 +449,7 @@ console.log(`User said: ${transcript}`)
|
|
|
476
449
|
const { text } = await voiceAgent.generate(transcript)
|
|
477
450
|
```
|
|
478
451
|
|
|
479
|
-
Visit the [Cloudflare Voice Reference](https://mastra.ai/
|
|
452
|
+
Visit the [Cloudflare Voice Reference](https://mastra.ai/integrations/voice/cloudflare) for more information on the Cloudflare voice provider.
|
|
480
453
|
|
|
481
454
|
**Deepgram**:
|
|
482
455
|
|
|
@@ -504,7 +477,7 @@ console.log(`User said: ${transcript}`)
|
|
|
504
477
|
const { text } = await voiceAgent.generate(transcript)
|
|
505
478
|
```
|
|
506
479
|
|
|
507
|
-
Visit the [Deepgram Voice Reference](https://mastra.ai/
|
|
480
|
+
Visit the [Deepgram Voice Reference](https://mastra.ai/integrations/voice/deepgram) for more information on the Deepgram voice provider.
|
|
508
481
|
|
|
509
482
|
**Inworld**:
|
|
510
483
|
|
|
@@ -532,7 +505,7 @@ console.log(`User said: ${transcript}`)
|
|
|
532
505
|
const { text } = await voiceAgent.generate(transcript)
|
|
533
506
|
```
|
|
534
507
|
|
|
535
|
-
Visit the [Inworld Voice Reference](https://mastra.ai/
|
|
508
|
+
Visit the [Inworld Voice Reference](https://mastra.ai/integrations/voice/inworld) for more information on the Inworld voice provider.
|
|
536
509
|
|
|
537
510
|
**Sarvam**:
|
|
538
511
|
|
|
@@ -560,7 +533,7 @@ console.log(`User said: ${transcript}`)
|
|
|
560
533
|
const { text } = await voiceAgent.generate(transcript)
|
|
561
534
|
```
|
|
562
535
|
|
|
563
|
-
Visit the [Sarvam Voice Reference](https://mastra.ai/
|
|
536
|
+
Visit the [Sarvam Voice Reference](https://mastra.ai/integrations/voice/sarvam) for more information on the Sarvam voice provider.
|
|
564
537
|
|
|
565
538
|
### Speech to Speech (STS)
|
|
566
539
|
|
|
@@ -594,7 +567,7 @@ const micStream = getMicrophoneStream()
|
|
|
594
567
|
await voiceAgent.voice.send(micStream)
|
|
595
568
|
```
|
|
596
569
|
|
|
597
|
-
Visit the [OpenAI Voice Reference](https://mastra.ai/
|
|
570
|
+
Visit the [OpenAI Voice Reference](https://mastra.ai/integrations/voice/openai) for more information on the OpenAI voice provider.
|
|
598
571
|
|
|
599
572
|
**Google**:
|
|
600
573
|
|
|
@@ -643,7 +616,7 @@ const micStream = getMicrophoneStream()
|
|
|
643
616
|
await voiceAgent.voice.send(micStream)
|
|
644
617
|
```
|
|
645
618
|
|
|
646
|
-
Visit the [Google Gemini Live Reference](https://mastra.ai/
|
|
619
|
+
Visit the [Google Gemini Live Reference](https://mastra.ai/integrations/voice/google) for more information on the Google Gemini Live voice provider.
|
|
647
620
|
|
|
648
621
|
**AWS Nova Sonic**:
|
|
649
622
|
|
|
@@ -686,7 +659,7 @@ const micStream = getMicrophoneStream()
|
|
|
686
659
|
await voiceAgent.voice.send(micStream)
|
|
687
660
|
```
|
|
688
661
|
|
|
689
|
-
Visit the [AWS Nova Sonic Reference](https://mastra.ai/
|
|
662
|
+
Visit the [AWS Nova Sonic Reference](https://mastra.ai/integrations/voice/aws-nova-sonic) for more information on the AWS Nova Sonic voice provider.
|
|
690
663
|
|
|
691
664
|
**Inworld Realtime**:
|
|
692
665
|
|
|
@@ -728,7 +701,7 @@ const micStream = getMicrophoneStream()
|
|
|
728
701
|
await voiceAgent.voice.send(micStream)
|
|
729
702
|
```
|
|
730
703
|
|
|
731
|
-
Visit the [Inworld Realtime Reference](https://mastra.ai/
|
|
704
|
+
Visit the [Inworld Realtime Reference](https://mastra.ai/integrations/voice/inworld) for more information on the Inworld Realtime voice provider.
|
|
732
705
|
|
|
733
706
|
**xAI**:
|
|
734
707
|
|
|
@@ -771,11 +744,11 @@ const micStream = getMicrophoneStream()
|
|
|
771
744
|
await voiceAgent.voice.send(micStream)
|
|
772
745
|
```
|
|
773
746
|
|
|
774
|
-
Visit the [xAI Realtime Voice Reference](https://mastra.ai/
|
|
747
|
+
Visit the [xAI Realtime Voice Reference](https://mastra.ai/integrations/voice/xai) for more information on the xAI voice provider.
|
|
775
748
|
|
|
776
749
|
### Realtime voice
|
|
777
750
|
|
|
778
|
-
Run live calls a user can talk over, in the browser or over the phone. Mastra hands the audio loop to LiveKit, which covers voice activity detection, semantic turn detection, and barge-in, while your agent generates each reply with its own model, tools, and memory. For setup and configuration options, check out [Realtime voice](https://mastra.ai/
|
|
751
|
+
Run live calls a user can talk over, in the browser or over the phone. Mastra hands the audio loop to LiveKit, which covers voice activity detection, semantic turn detection, and barge-in, while your agent generates each reply with its own model, tools, and memory. For setup and configuration options, check out [Realtime voice](https://mastra.ai/integrations/voice/livekit).
|
|
779
752
|
|
|
780
753
|
## Voice configuration
|
|
781
754
|
|
|
@@ -802,7 +775,7 @@ const voice = new OpenAIVoice({
|
|
|
802
775
|
})
|
|
803
776
|
```
|
|
804
777
|
|
|
805
|
-
Visit the [OpenAI Voice Reference](https://mastra.ai/
|
|
778
|
+
Visit the [OpenAI Voice Reference](https://mastra.ai/integrations/voice/openai) for more information on the OpenAI voice provider.
|
|
806
779
|
|
|
807
780
|
**Azure**:
|
|
808
781
|
|
|
@@ -827,7 +800,7 @@ const voice = new AzureVoice({
|
|
|
827
800
|
})
|
|
828
801
|
```
|
|
829
802
|
|
|
830
|
-
Visit the [Azure Voice Reference](https://mastra.ai/
|
|
803
|
+
Visit the [Azure Voice Reference](https://mastra.ai/integrations/voice/azure) for more information on the Azure voice provider.
|
|
831
804
|
|
|
832
805
|
**ElevenLabs**:
|
|
833
806
|
|
|
@@ -845,25 +818,7 @@ const voice = new ElevenLabsVoice({
|
|
|
845
818
|
})
|
|
846
819
|
```
|
|
847
820
|
|
|
848
|
-
Visit the [ElevenLabs Voice Reference](https://mastra.ai/
|
|
849
|
-
|
|
850
|
-
**PlayAI**:
|
|
851
|
-
|
|
852
|
-
```typescript
|
|
853
|
-
// PlayAI Voice Configuration
|
|
854
|
-
const voice = new PlayAIVoice({
|
|
855
|
-
speechModel: {
|
|
856
|
-
name: 'playai-voice', // Example model name
|
|
857
|
-
speaker: 'emma', // Example speaker name
|
|
858
|
-
apiKey: process.env.PLAYAI_API_KEY,
|
|
859
|
-
language: 'en-US', // Language code
|
|
860
|
-
speed: 1.0, // Speech speed
|
|
861
|
-
},
|
|
862
|
-
// PlayAI may not have a separate listening model
|
|
863
|
-
})
|
|
864
|
-
```
|
|
865
|
-
|
|
866
|
-
Visit the [PlayAI Voice Reference](https://mastra.ai/reference/voice/playai) for more information on the PlayAI voice provider.
|
|
821
|
+
Visit the [ElevenLabs Voice Reference](https://mastra.ai/integrations/voice/elevenlabs) for more information on the ElevenLabs voice provider.
|
|
867
822
|
|
|
868
823
|
**Google**:
|
|
869
824
|
|
|
@@ -884,7 +839,7 @@ const voice = new GoogleVoice({
|
|
|
884
839
|
})
|
|
885
840
|
```
|
|
886
841
|
|
|
887
|
-
Visit the [Google Voice Reference](https://mastra.ai/
|
|
842
|
+
Visit the [Google Voice Reference](https://mastra.ai/integrations/voice/google) for more information on the Google voice provider.
|
|
888
843
|
|
|
889
844
|
**Cloudflare**:
|
|
890
845
|
|
|
@@ -902,7 +857,7 @@ const voice = new CloudflareVoice({
|
|
|
902
857
|
})
|
|
903
858
|
```
|
|
904
859
|
|
|
905
|
-
Visit the [Cloudflare Voice Reference](https://mastra.ai/
|
|
860
|
+
Visit the [Cloudflare Voice Reference](https://mastra.ai/integrations/voice/cloudflare) for more information on the Cloudflare voice provider.
|
|
906
861
|
|
|
907
862
|
**Deepgram**:
|
|
908
863
|
|
|
@@ -923,7 +878,7 @@ const voice = new DeepgramVoice({
|
|
|
923
878
|
})
|
|
924
879
|
```
|
|
925
880
|
|
|
926
|
-
Visit the [Deepgram Voice Reference](https://mastra.ai/
|
|
881
|
+
Visit the [Deepgram Voice Reference](https://mastra.ai/integrations/voice/deepgram) for more information on the Deepgram voice provider.
|
|
927
882
|
|
|
928
883
|
**Inworld**:
|
|
929
884
|
|
|
@@ -951,7 +906,7 @@ const audioStream = await voice.speak('Hello!', {
|
|
|
951
906
|
})
|
|
952
907
|
```
|
|
953
908
|
|
|
954
|
-
Visit the [Inworld Voice Reference](https://mastra.ai/
|
|
909
|
+
Visit the [Inworld Voice Reference](https://mastra.ai/integrations/voice/inworld) for more information on the Inworld voice provider.
|
|
955
910
|
|
|
956
911
|
**Speechify**:
|
|
957
912
|
|
|
@@ -969,7 +924,7 @@ const voice = new SpeechifyVoice({
|
|
|
969
924
|
})
|
|
970
925
|
```
|
|
971
926
|
|
|
972
|
-
Visit the [Speechify Voice Reference](https://mastra.ai/
|
|
927
|
+
Visit the [Speechify Voice Reference](https://mastra.ai/integrations/voice/speechify) for more information on the Speechify voice provider.
|
|
973
928
|
|
|
974
929
|
**Sarvam**:
|
|
975
930
|
|
|
@@ -989,7 +944,7 @@ const voice = new SarvamVoice({
|
|
|
989
944
|
})
|
|
990
945
|
```
|
|
991
946
|
|
|
992
|
-
Visit the [Sarvam Voice Reference](https://mastra.ai/
|
|
947
|
+
Visit the [Sarvam Voice Reference](https://mastra.ai/integrations/voice/sarvam) for more information on the Sarvam voice provider.
|
|
993
948
|
|
|
994
949
|
**Murf**:
|
|
995
950
|
|
|
@@ -1006,7 +961,7 @@ const voice = new MurfVoice({
|
|
|
1006
961
|
})
|
|
1007
962
|
```
|
|
1008
963
|
|
|
1009
|
-
Visit the [Murf Voice Reference](https://mastra.ai/
|
|
964
|
+
Visit the [Murf Voice Reference](https://mastra.ai/integrations/voice/murf) for more information on the Murf voice provider.
|
|
1010
965
|
|
|
1011
966
|
**OpenAI Realtime**:
|
|
1012
967
|
|
|
@@ -1027,7 +982,7 @@ const voice = new OpenAIRealtimeVoice({
|
|
|
1027
982
|
})
|
|
1028
983
|
```
|
|
1029
984
|
|
|
1030
|
-
For more information on the OpenAI Realtime voice provider, refer to the [OpenAI Realtime Voice Reference](https://mastra.ai/
|
|
985
|
+
For more information on the OpenAI Realtime voice provider, refer to the [OpenAI Realtime Voice Reference](https://mastra.ai/integrations/voice/openai).
|
|
1031
986
|
|
|
1032
987
|
**xAI Realtime**:
|
|
1033
988
|
|
|
@@ -1059,7 +1014,7 @@ const voice = new XAIRealtimeVoice({
|
|
|
1059
1014
|
})
|
|
1060
1015
|
```
|
|
1061
1016
|
|
|
1062
|
-
Visit the [xAI Realtime Voice Reference](https://mastra.ai/
|
|
1017
|
+
Visit the [xAI Realtime Voice Reference](https://mastra.ai/integrations/voice/xai) for more information on the xAI realtime voice provider.
|
|
1063
1018
|
|
|
1064
1019
|
**Google Gemini Live**:
|
|
1065
1020
|
|
|
@@ -1075,7 +1030,7 @@ const voice = new GeminiLiveVoice({
|
|
|
1075
1030
|
})
|
|
1076
1031
|
```
|
|
1077
1032
|
|
|
1078
|
-
Visit the [Google Gemini Live Reference](https://mastra.ai/
|
|
1033
|
+
Visit the [Google Gemini Live Reference](https://mastra.ai/integrations/voice/google) for more information on the Google Gemini Live voice provider.
|
|
1079
1034
|
|
|
1080
1035
|
**AWS Nova Sonic**:
|
|
1081
1036
|
|
|
@@ -1097,7 +1052,7 @@ const voice = new NovaSonicVoice({
|
|
|
1097
1052
|
})
|
|
1098
1053
|
```
|
|
1099
1054
|
|
|
1100
|
-
Visit the [AWS Nova Sonic Reference](https://mastra.ai/
|
|
1055
|
+
Visit the [AWS Nova Sonic Reference](https://mastra.ai/integrations/voice/aws-nova-sonic) for more information on the AWS Nova Sonic voice provider.
|
|
1101
1056
|
|
|
1102
1057
|
**Inworld Realtime**:
|
|
1103
1058
|
|
|
@@ -1117,7 +1072,7 @@ const voice = new InworldRealtimeVoice({
|
|
|
1117
1072
|
})
|
|
1118
1073
|
```
|
|
1119
1074
|
|
|
1120
|
-
Visit the [Inworld Realtime Reference](https://mastra.ai/
|
|
1075
|
+
Visit the [Inworld Realtime Reference](https://mastra.ai/integrations/voice/inworld) for more information on the Inworld Realtime voice provider.
|
|
1121
1076
|
|
|
1122
1077
|
**AI SDK**:
|
|
1123
1078
|
|
|
@@ -1145,13 +1100,13 @@ const voiceAgent = new Agent({
|
|
|
1145
1100
|
|
|
1146
1101
|
### Using multiple voice providers
|
|
1147
1102
|
|
|
1148
|
-
This example demonstrates how to create and use two different voice providers in Mastra: OpenAI for speech-to-text (STT) and
|
|
1103
|
+
This example demonstrates how to create and use two different voice providers in Mastra: OpenAI for speech-to-text (STT) and ElevenLabs for text-to-speech (TTS).
|
|
1149
1104
|
|
|
1150
1105
|
Start by creating instances of the voice providers with any necessary configuration.
|
|
1151
1106
|
|
|
1152
1107
|
```typescript
|
|
1153
1108
|
import { OpenAIVoice } from '@mastra/voice-openai'
|
|
1154
|
-
import {
|
|
1109
|
+
import { ElevenLabsVoice } from '@mastra/voice-elevenlabs'
|
|
1155
1110
|
import { CompositeVoice } from '@mastra/core/voice'
|
|
1156
1111
|
import { playAudio, getMicrophoneStream } from '@mastra/node-audio'
|
|
1157
1112
|
|
|
@@ -1163,13 +1118,8 @@ const input = new OpenAIVoice({
|
|
|
1163
1118
|
},
|
|
1164
1119
|
})
|
|
1165
1120
|
|
|
1166
|
-
// Initialize
|
|
1167
|
-
const output = new
|
|
1168
|
-
speechModel: {
|
|
1169
|
-
name: 'playai-voice',
|
|
1170
|
-
apiKey: process.env.PLAYAI_API_KEY,
|
|
1171
|
-
},
|
|
1172
|
-
})
|
|
1121
|
+
// Initialize ElevenLabs voice for TTS
|
|
1122
|
+
const output = new ElevenLabsVoice()
|
|
1173
1123
|
|
|
1174
1124
|
// Combine the providers using CompositeVoice
|
|
1175
1125
|
const voice = new CompositeVoice({
|
|
@@ -1228,12 +1178,12 @@ You can also mix AI SDK models with Mastra providers:
|
|
|
1228
1178
|
|
|
1229
1179
|
```typescript
|
|
1230
1180
|
import { CompositeVoice } from '@mastra/core/voice'
|
|
1231
|
-
import {
|
|
1181
|
+
import { ElevenLabsVoice } from '@mastra/voice-elevenlabs'
|
|
1232
1182
|
import { groq } from '@ai-sdk/groq'
|
|
1233
1183
|
|
|
1234
1184
|
const voice = new CompositeVoice({
|
|
1235
1185
|
input: groq.transcription('whisper-large-v3'), // AI SDK for STT
|
|
1236
|
-
output: new
|
|
1186
|
+
output: new ElevenLabsVoice(), // Mastra provider for TTS
|
|
1237
1187
|
})
|
|
1238
1188
|
```
|
|
1239
1189
|
|
|
@@ -1243,15 +1193,14 @@ For more information on the CompositeVoice, refer to the [CompositeVoice Referen
|
|
|
1243
1193
|
|
|
1244
1194
|
- [CompositeVoice](https://mastra.ai/reference/voice/composite-voice)
|
|
1245
1195
|
- [MastraVoice](https://mastra.ai/reference/voice/mastra-voice)
|
|
1246
|
-
- [OpenAI Voice](https://mastra.ai/
|
|
1247
|
-
- [OpenAI Realtime Voice](https://mastra.ai/
|
|
1248
|
-
- [xAI Realtime Voice](https://mastra.ai/
|
|
1249
|
-
- [Azure Voice](https://mastra.ai/
|
|
1250
|
-
- [Google Voice](https://mastra.ai/
|
|
1251
|
-
- [Google Gemini Live Voice](https://mastra.ai/
|
|
1252
|
-
- [AWS Nova Sonic Voice](https://mastra.ai/
|
|
1253
|
-
- [Deepgram Voice](https://mastra.ai/
|
|
1254
|
-
- [Inworld Voice](https://mastra.ai/
|
|
1255
|
-
- [LiveKit](https://mastra.ai/
|
|
1256
|
-
- [PlayAI Voice](https://mastra.ai/reference/voice/playai)
|
|
1196
|
+
- [OpenAI Voice](https://mastra.ai/integrations/voice/openai)
|
|
1197
|
+
- [OpenAI Realtime Voice](https://mastra.ai/integrations/voice/openai)
|
|
1198
|
+
- [xAI Realtime Voice](https://mastra.ai/integrations/voice/xai)
|
|
1199
|
+
- [Azure Voice](https://mastra.ai/integrations/voice/azure)
|
|
1200
|
+
- [Google Voice](https://mastra.ai/integrations/voice/google)
|
|
1201
|
+
- [Google Gemini Live Voice](https://mastra.ai/integrations/voice/google)
|
|
1202
|
+
- [AWS Nova Sonic Voice](https://mastra.ai/integrations/voice/aws-nova-sonic)
|
|
1203
|
+
- [Deepgram Voice](https://mastra.ai/integrations/voice/deepgram)
|
|
1204
|
+
- [Inworld Voice](https://mastra.ai/integrations/voice/inworld)
|
|
1205
|
+
- [LiveKit](https://mastra.ai/integrations/voice/livekit)
|
|
1257
1206
|
- [Voice Examples](https://github.com/mastra-ai/voice-examples)
|
package/dist/docs/references/{reference-voice-google-gemini-live.md → integrations-voice-google.md}
RENAMED
|
@@ -1,10 +1,285 @@
|
|
|
1
1
|
> Discover all available pages from the documentation index: https://mastra.ai/llms.txt
|
|
2
2
|
|
|
3
|
-
# Google
|
|
3
|
+
# Google
|
|
4
|
+
|
|
5
|
+
## Speech
|
|
6
|
+
|
|
7
|
+
The Google Voice implementation in Mastra provides both text-to-speech (TTS) and speech-to-text (STT) capabilities using Google Cloud services. It supports multiple voices, languages, advanced audio configuration options, and both standard API key authentication and Vertex AI mode for enterprise deployments.
|
|
8
|
+
|
|
9
|
+
### Usage example
|
|
10
|
+
|
|
11
|
+
```typescript
|
|
12
|
+
import { GoogleVoice } from '@mastra/voice-google'
|
|
13
|
+
|
|
14
|
+
// Initialize with default configuration (uses GOOGLE_API_KEY environment variable)
|
|
15
|
+
const voice = new GoogleVoice()
|
|
16
|
+
|
|
17
|
+
// Text-to-Speech (plain text)
|
|
18
|
+
const audioStream = await voice.speak('Hello, world!', {
|
|
19
|
+
languageCode: 'en-US',
|
|
20
|
+
audioConfig: {
|
|
21
|
+
audioEncoding: 'LINEAR16',
|
|
22
|
+
},
|
|
23
|
+
})
|
|
24
|
+
|
|
25
|
+
// Text-to-Speech with SSML
|
|
26
|
+
const ssmlStream = await voice.speak('ignored', {
|
|
27
|
+
input: {
|
|
28
|
+
ssml: '<speak>Take <say-as interpret-as="unit">5 mg</say-as> daily.</speak>',
|
|
29
|
+
},
|
|
30
|
+
})
|
|
31
|
+
|
|
32
|
+
// Text-to-Speech with Gemini-TTS model
|
|
33
|
+
const geminiStream = await voice.speak('Hello from Gemini TTS!', {
|
|
34
|
+
voice: { name: 'Kore', modelName: 'gemini-2.5-flash-preview-tts' },
|
|
35
|
+
input: { prompt: 'Warm, calm tone.' },
|
|
36
|
+
})
|
|
37
|
+
|
|
38
|
+
// Speech-to-Text
|
|
39
|
+
const transcript = await voice.listen(audioStream, {
|
|
40
|
+
config: {
|
|
41
|
+
encoding: 'LINEAR16',
|
|
42
|
+
languageCode: 'en-US',
|
|
43
|
+
},
|
|
44
|
+
})
|
|
45
|
+
|
|
46
|
+
// Get available voices for a specific language
|
|
47
|
+
const voices = await voice.getSpeakers({ languageCode: 'en-US' })
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
### Constructor parameters
|
|
51
|
+
|
|
52
|
+
**speechModel** (`GoogleModelConfig`): Configuration for text-to-speech functionality (Default: `{ apiKey: process.env.GOOGLE_API_KEY }`)
|
|
53
|
+
|
|
54
|
+
**speechModel.apiKey** (`string`): Google Cloud API key. Falls back to GOOGLE\_API\_KEY environment variable. Not used when vertexAI is true.
|
|
55
|
+
|
|
56
|
+
**speechModel.keyFilename** (`string`): Path to service account JSON key file. Falls back to GOOGLE\_APPLICATION\_CREDENTIALS environment variable.
|
|
57
|
+
|
|
58
|
+
**speechModel.credentials** (`object`): In-memory service account credentials object with client\_email and private\_key properties.
|
|
59
|
+
|
|
60
|
+
**listeningModel** (`GoogleModelConfig`): Configuration for speech-to-text functionality (Default: `{ apiKey: process.env.GOOGLE_API_KEY }`)
|
|
61
|
+
|
|
62
|
+
**listeningModel.apiKey** (`string`): Google Cloud API key. Falls back to GOOGLE\_API\_KEY environment variable. Not used when vertexAI is true.
|
|
63
|
+
|
|
64
|
+
**listeningModel.keyFilename** (`string`): Path to service account JSON key file. Falls back to GOOGLE\_APPLICATION\_CREDENTIALS environment variable.
|
|
65
|
+
|
|
66
|
+
**listeningModel.credentials** (`object`): In-memory service account credentials object with client\_email and private\_key properties.
|
|
67
|
+
|
|
68
|
+
**speaker** (`string`): Default voice ID to use for text-to-speech (Default: `'en-US-Casual-K'`)
|
|
69
|
+
|
|
70
|
+
**vertexAI** (`boolean`): Enable Vertex AI mode for enterprise deployments. Uses project-based authentication instead of API keys. Requires 'project' to be set. (Default: `false`)
|
|
71
|
+
|
|
72
|
+
**project** (`string`): Google Cloud project ID (required when vertexAI is true). Falls back to GOOGLE\_CLOUD\_PROJECT environment variable.
|
|
73
|
+
|
|
74
|
+
**location** (`string`): Google Cloud region for Vertex AI. Falls back to GOOGLE\_CLOUD\_LOCATION environment variable. (Default: `'us-central1'`)
|
|
75
|
+
|
|
76
|
+
### Methods
|
|
77
|
+
|
|
78
|
+
#### `speak()`
|
|
79
|
+
|
|
80
|
+
Converts text to speech using Google Cloud Text-to-Speech service.
|
|
81
|
+
|
|
82
|
+
**input** (`string | NodeJS.ReadableStream`): Text to convert to speech. If a stream is provided, it will be converted to text first.
|
|
83
|
+
|
|
84
|
+
**options** (`object`): Speech synthesis options
|
|
85
|
+
|
|
86
|
+
**options.speaker** (`string`): Voice ID to use for this request.
|
|
87
|
+
|
|
88
|
+
**options.languageCode** (`string`): Language code for the voice (e.g., 'en-US'). Defaults to the language code derived from the speaker ID, or 'en-US'.
|
|
89
|
+
|
|
90
|
+
**options.input** (`ISynthesizeSpeechRequest['input']`): Rich input object passed through to the Google Cloud TTS API. Supports ssml, markup, prompt (Gemini-TTS style steering), customPronunciations, and multiSpeakerMarkup. When provided without text, ssml, markup, or multiSpeakerMarkup, the positional input argument is used as the text field automatically.
|
|
91
|
+
|
|
92
|
+
**options.voice** (`ISynthesizeSpeechRequest['voice']`): Voice configuration merged on top of defaults (name and languageCode). Supports modelName (e.g., 'gemini-2.5-flash-preview-tts') and multiSpeakerVoiceConfig.
|
|
93
|
+
|
|
94
|
+
**options.audioConfig** (`ISynthesizeSpeechRequest['audioConfig']`): Audio configuration options from Google Cloud Text-to-Speech API.
|
|
95
|
+
|
|
96
|
+
Returns: `Promise<NodeJS.ReadableStream>`
|
|
97
|
+
|
|
98
|
+
#### `listen()`
|
|
99
|
+
|
|
100
|
+
Converts speech to text using Google Cloud Speech-to-Text service. Supports both v1 (default) and v2 APIs. The v2 API adds support for AAC-in-MP4 audio (iOS Safari) via auto-decoding.
|
|
101
|
+
|
|
102
|
+
##### v1 (default)
|
|
103
|
+
|
|
104
|
+
**audioStream** (`NodeJS.ReadableStream`): Audio stream to transcribe
|
|
105
|
+
|
|
106
|
+
**options** (`GoogleListenOptionsV1`): v1 recognition options
|
|
107
|
+
|
|
108
|
+
**options.config** (`IRecognitionConfig`): v1 recognition configuration from Google Cloud Speech-to-Text API
|
|
109
|
+
|
|
110
|
+
##### v2
|
|
111
|
+
|
|
112
|
+
Pass `v2: true` to use the Cloud Speech-to-Text v2 API, which supports additional audio formats like AAC-in-MP4 (iOS Safari).
|
|
113
|
+
|
|
114
|
+
```typescript
|
|
115
|
+
const transcript = await voice.listen(iosSafariAacStream, {
|
|
116
|
+
v2: true,
|
|
117
|
+
config: {
|
|
118
|
+
autoDecodingConfig: {},
|
|
119
|
+
},
|
|
120
|
+
})
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
**audioStream** (`NodeJS.ReadableStream`): Audio stream to transcribe
|
|
124
|
+
|
|
125
|
+
**options** (`GoogleListenOptionsV2`): v2 recognition options
|
|
126
|
+
|
|
127
|
+
**options.v2** (`true`): Enables the v2 API path
|
|
128
|
+
|
|
129
|
+
**options.config** (`v2.IRecognitionConfig`): v2 recognition configuration. Defaults to auto-decoding with languageCodes: \['en-US'] and model: 'long'. Set autoDecodingConfig: {} to auto-detect the audio format, or use explicitDecodingConfig to specify an encoding like MP4\_AAC, M4A\_AAC, or MOV\_AAC.
|
|
130
|
+
|
|
131
|
+
**options.recognizer** (`string`): v2 recognizer resource path. Defaults to projects/{project}/locations/global/recognizers/\_ where {project} is resolved from the constructor project option, GOOGLE\_CLOUD\_PROJECT, or the client's default project.
|
|
132
|
+
|
|
133
|
+
Returns: `Promise<string>`
|
|
134
|
+
|
|
135
|
+
#### `getSpeakers()`
|
|
136
|
+
|
|
137
|
+
Returns an array of available voice options, where each node contains:
|
|
138
|
+
|
|
139
|
+
**voiceId** (`string`): Unique identifier for the voice
|
|
140
|
+
|
|
141
|
+
**languageCodes** (`string[]`): List of language codes supported by this voice
|
|
142
|
+
|
|
143
|
+
#### `isUsingVertexAI()`
|
|
144
|
+
|
|
145
|
+
Checks if Vertex AI mode is enabled.
|
|
146
|
+
|
|
147
|
+
Returns: `boolean` - `true` if using Vertex AI, `false` otherwise
|
|
148
|
+
|
|
149
|
+
#### `getProject()`
|
|
150
|
+
|
|
151
|
+
Gets the configured Google Cloud project ID.
|
|
152
|
+
|
|
153
|
+
Returns: `string | undefined` - The project ID or `undefined` if not set
|
|
154
|
+
|
|
155
|
+
#### `getLocation()`
|
|
156
|
+
|
|
157
|
+
Gets the configured Google Cloud location/region.
|
|
158
|
+
|
|
159
|
+
Returns: `string` - The location (default: `'us-central1'`)
|
|
160
|
+
|
|
161
|
+
### Authentication
|
|
162
|
+
|
|
163
|
+
The Google Voice provider supports two authentication methods:
|
|
164
|
+
|
|
165
|
+
#### Standard Mode (API Key)
|
|
166
|
+
|
|
167
|
+
Uses a Google Cloud API key for authentication. Suitable for development and basic use cases.
|
|
168
|
+
|
|
169
|
+
```typescript
|
|
170
|
+
// Using environment variable (GOOGLE_API_KEY)
|
|
171
|
+
const voice = new GoogleVoice()
|
|
172
|
+
|
|
173
|
+
// Using explicit API key
|
|
174
|
+
const voice = new GoogleVoice({
|
|
175
|
+
speechModel: { apiKey: 'your-api-key' },
|
|
176
|
+
listeningModel: { apiKey: 'your-api-key' },
|
|
177
|
+
speaker: 'en-US-Casual-K',
|
|
178
|
+
})
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
#### Vertex AI Mode (Service Account)
|
|
182
|
+
|
|
183
|
+
Uses Google Cloud project-based authentication with service accounts. Recommended for production and enterprise deployments.
|
|
184
|
+
|
|
185
|
+
**Benefits:**
|
|
186
|
+
|
|
187
|
+
- Better security (no API keys in code)
|
|
188
|
+
- IAM-based access control
|
|
189
|
+
- Project-level billing and quotas
|
|
190
|
+
- Audit logging
|
|
191
|
+
- Enterprise features
|
|
192
|
+
|
|
193
|
+
**Configuration Options:**
|
|
194
|
+
|
|
195
|
+
```typescript
|
|
196
|
+
// Using Application Default Credentials (ADC)
|
|
197
|
+
// Set GOOGLE_APPLICATION_CREDENTIALS and GOOGLE_CLOUD_PROJECT env vars
|
|
198
|
+
const voice = new GoogleVoice({
|
|
199
|
+
vertexAI: true,
|
|
200
|
+
project: 'your-gcp-project',
|
|
201
|
+
location: 'us-central1', // Optional, defaults to 'us-central1'
|
|
202
|
+
})
|
|
203
|
+
|
|
204
|
+
// Using service account key file
|
|
205
|
+
const voice = new GoogleVoice({
|
|
206
|
+
vertexAI: true,
|
|
207
|
+
project: 'your-gcp-project',
|
|
208
|
+
speechModel: {
|
|
209
|
+
keyFilename: '/path/to/service-account.json',
|
|
210
|
+
},
|
|
211
|
+
listeningModel: {
|
|
212
|
+
keyFilename: '/path/to/service-account.json',
|
|
213
|
+
},
|
|
214
|
+
})
|
|
215
|
+
|
|
216
|
+
// Using in-memory credentials
|
|
217
|
+
const voice = new GoogleVoice({
|
|
218
|
+
vertexAI: true,
|
|
219
|
+
project: 'your-gcp-project',
|
|
220
|
+
speechModel: {
|
|
221
|
+
credentials: {
|
|
222
|
+
client_email: 'service-account@project.iam.gserviceaccount.com',
|
|
223
|
+
private_key: '-----BEGIN PRIVATE KEY-----\n...\n-----END PRIVATE KEY-----',
|
|
224
|
+
},
|
|
225
|
+
},
|
|
226
|
+
})
|
|
227
|
+
```
|
|
228
|
+
|
|
229
|
+
##### Required Permissions
|
|
230
|
+
|
|
231
|
+
##### IAM Roles
|
|
232
|
+
|
|
233
|
+
For Text-to-Speech:
|
|
234
|
+
|
|
235
|
+
- `roles/texttospeech.admin` - Text-to-Speech Admin (full access)
|
|
236
|
+
- `roles/texttospeech.editor` - Text-to-Speech Editor (create and manage)
|
|
237
|
+
- `roles/texttospeech.viewer` - Text-to-Speech Viewer (read-only)
|
|
238
|
+
|
|
239
|
+
For Speech-to-Text:
|
|
240
|
+
|
|
241
|
+
- `roles/speech.client` - Speech-to-Text Client
|
|
242
|
+
|
|
243
|
+
##### OAuth Scopes
|
|
244
|
+
|
|
245
|
+
For synchronous Text-to-Speech synthesis:
|
|
246
|
+
|
|
247
|
+
- `https://www.googleapis.com/auth/cloud-platform` - Full access to Google Cloud Platform services
|
|
248
|
+
|
|
249
|
+
For long-audio Text-to-Speech operations:
|
|
250
|
+
|
|
251
|
+
- `locations.longAudioSynthesize` - Create long-audio synthesis operations
|
|
252
|
+
- `operations.get` - Get operation status
|
|
253
|
+
- `operations.list` - List operations
|
|
254
|
+
|
|
255
|
+
### Important notes
|
|
256
|
+
|
|
257
|
+
1. **Authentication**: Either a Google Cloud API key (standard mode) or service account credentials (Vertex AI mode) is required.
|
|
258
|
+
|
|
259
|
+
2. **Environment Variables**:
|
|
260
|
+
|
|
261
|
+
- `GOOGLE_API_KEY` - API key for standard mode
|
|
262
|
+
- `GOOGLE_CLOUD_PROJECT` - Project ID for Vertex AI mode
|
|
263
|
+
- `GOOGLE_CLOUD_LOCATION` - Location for Vertex AI mode (defaults to 'us-central1')
|
|
264
|
+
- `GOOGLE_APPLICATION_CREDENTIALS` - Path to service account key file
|
|
265
|
+
|
|
266
|
+
3. The default voice is set to `'en-US-Casual-K'`.
|
|
267
|
+
|
|
268
|
+
4. Both text-to-speech and speech-to-text services use LINEAR16 as the default audio encoding.
|
|
269
|
+
|
|
270
|
+
5. The `speak()` method supports advanced audio configuration through the Google Cloud Text-to-Speech API.
|
|
271
|
+
|
|
272
|
+
6. The `listen()` method supports various recognition configurations through the Google Cloud Speech-to-Text API.
|
|
273
|
+
|
|
274
|
+
7. Available voices can be filtered by language code using the `getSpeakers()` method.
|
|
275
|
+
|
|
276
|
+
8. Vertex AI mode provides enterprise features including IAM control, audit logs, and project-level billing.
|
|
277
|
+
|
|
278
|
+
## Gemini Live
|
|
4
279
|
|
|
5
280
|
The GeminiLiveVoice class provides real-time voice interaction capabilities using Google's Gemini Live API. It supports bidirectional audio streaming, tool calling, session management, and both standard Google API and Vertex AI authentication methods.
|
|
6
281
|
|
|
7
|
-
|
|
282
|
+
### Usage example
|
|
8
283
|
|
|
9
284
|
```typescript
|
|
10
285
|
import { GeminiLiveVoice } from '@mastra/voice-google-gemini-live'
|
|
@@ -88,9 +363,9 @@ await voice.disconnect()
|
|
|
88
363
|
voice.close()
|
|
89
364
|
```
|
|
90
365
|
|
|
91
|
-
|
|
366
|
+
### Configuration
|
|
92
367
|
|
|
93
|
-
|
|
368
|
+
#### Constructor options
|
|
94
369
|
|
|
95
370
|
**apiKey** (`string`): Google API key for Gemini API authentication. Required unless using Vertex AI.
|
|
96
371
|
|
|
@@ -122,9 +397,9 @@ voice.close()
|
|
|
122
397
|
|
|
123
398
|
**debug** (`boolean`): Enable debug logging for troubleshooting. (Default: `false`)
|
|
124
399
|
|
|
125
|
-
|
|
400
|
+
### Methods
|
|
126
401
|
|
|
127
|
-
|
|
402
|
+
#### `connect()`
|
|
128
403
|
|
|
129
404
|
Establishes a connection to the Gemini Live API. Must be called before using speak, listen, or send methods.
|
|
130
405
|
|
|
@@ -132,7 +407,7 @@ Establishes a connection to the Gemini Live API. Must be called before using spe
|
|
|
132
407
|
|
|
133
408
|
**returns** (`Promise<void>`): Promise that resolves when the connection is established.
|
|
134
409
|
|
|
135
|
-
|
|
410
|
+
#### `speak()`
|
|
136
411
|
|
|
137
412
|
Converts text to speech and sends it to the model. Can accept either a string or a readable stream as input.
|
|
138
413
|
|
|
@@ -148,7 +423,7 @@ Converts text to speech and sends it to the model. Can accept either a string or
|
|
|
148
423
|
|
|
149
424
|
Returns: `Promise<void>` (responses are emitted via `speaker` and `writing` events)
|
|
150
425
|
|
|
151
|
-
|
|
426
|
+
#### `sendContext()`
|
|
152
427
|
|
|
153
428
|
Sends conversation history into the live session without triggering a model response. Use this to seed prior turns (e.g. from Mastra Memory) on a cold connect so the model has context before the user speaks.
|
|
154
429
|
|
|
@@ -170,7 +445,7 @@ await voice.send(micStream)
|
|
|
170
445
|
|
|
171
446
|
Returns: `Promise<void>`
|
|
172
447
|
|
|
173
|
-
|
|
448
|
+
#### `listen()`
|
|
174
449
|
|
|
175
450
|
Processes audio input for speech recognition. Takes a readable stream of audio data and returns the transcribed text.
|
|
176
451
|
|
|
@@ -180,7 +455,7 @@ Processes audio input for speech recognition. Takes a readable stream of audio d
|
|
|
180
455
|
|
|
181
456
|
Returns: `Promise<string>` - The transcribed text
|
|
182
457
|
|
|
183
|
-
|
|
458
|
+
#### `send()`
|
|
184
459
|
|
|
185
460
|
Streams audio data in real-time to the Gemini service for continuous audio streaming scenarios like live microphone input.
|
|
186
461
|
|
|
@@ -188,7 +463,7 @@ Streams audio data in real-time to the Gemini service for continuous audio strea
|
|
|
188
463
|
|
|
189
464
|
Returns: `Promise<void>`
|
|
190
465
|
|
|
191
|
-
|
|
466
|
+
#### `updateSessionConfig()`
|
|
192
467
|
|
|
193
468
|
Updates the session configuration at runtime. This can modify voice settings and speaker selection. It can also modify other runtime configurations.
|
|
194
469
|
|
|
@@ -196,7 +471,7 @@ Updates the session configuration at runtime. This can modify voice settings and
|
|
|
196
471
|
|
|
197
472
|
Returns: `Promise<void>`
|
|
198
473
|
|
|
199
|
-
|
|
474
|
+
#### `addTools()`
|
|
200
475
|
|
|
201
476
|
Adds a set of tools to the voice instance. Tools allow the model to perform additional actions during conversations. When GeminiLiveVoice is added to an Agent, any tools configured for the Agent will automatically be available to the voice interface.
|
|
202
477
|
|
|
@@ -204,7 +479,7 @@ Adds a set of tools to the voice instance. Tools allow the model to perform addi
|
|
|
204
479
|
|
|
205
480
|
Returns: `void`
|
|
206
481
|
|
|
207
|
-
|
|
482
|
+
#### `addInstructions()`
|
|
208
483
|
|
|
209
484
|
Adds or updates system instructions for the model.
|
|
210
485
|
|
|
@@ -212,7 +487,7 @@ Adds or updates system instructions for the model.
|
|
|
212
487
|
|
|
213
488
|
Returns: `void`
|
|
214
489
|
|
|
215
|
-
|
|
490
|
+
#### `answer()`
|
|
216
491
|
|
|
217
492
|
Triggers a response from the model. This method is primarily used internally when integrated with an Agent.
|
|
218
493
|
|
|
@@ -220,25 +495,25 @@ Triggers a response from the model. This method is primarily used internally whe
|
|
|
220
495
|
|
|
221
496
|
Returns: `Promise<void>`
|
|
222
497
|
|
|
223
|
-
|
|
498
|
+
#### `getSpeakers()`
|
|
224
499
|
|
|
225
500
|
Returns a list of available voice speakers for the Gemini Live API.
|
|
226
501
|
|
|
227
502
|
Returns: `Promise<Array<{ voiceId: string; description?: string }>>`
|
|
228
503
|
|
|
229
|
-
|
|
504
|
+
#### `disconnect()`
|
|
230
505
|
|
|
231
506
|
Disconnects from the Gemini Live session and cleans up resources. This is the async method that properly handles cleanup.
|
|
232
507
|
|
|
233
508
|
Returns: `Promise<void>`
|
|
234
509
|
|
|
235
|
-
|
|
510
|
+
#### `close()`
|
|
236
511
|
|
|
237
512
|
Synchronous wrapper for disconnect(). Calls disconnect() internally without awaiting.
|
|
238
513
|
|
|
239
514
|
Returns: `void`
|
|
240
515
|
|
|
241
|
-
|
|
516
|
+
#### `on()`
|
|
242
517
|
|
|
243
518
|
Registers an event listener for voice events.
|
|
244
519
|
|
|
@@ -248,7 +523,7 @@ Registers an event listener for voice events.
|
|
|
248
523
|
|
|
249
524
|
Returns: `void`
|
|
250
525
|
|
|
251
|
-
|
|
526
|
+
#### `off()`
|
|
252
527
|
|
|
253
528
|
Removes a previously registered event listener.
|
|
254
529
|
|
|
@@ -258,7 +533,7 @@ Removes a previously registered event listener.
|
|
|
258
533
|
|
|
259
534
|
Returns: `void`
|
|
260
535
|
|
|
261
|
-
|
|
536
|
+
### Events
|
|
262
537
|
|
|
263
538
|
The GeminiLiveVoice class emits the following events:
|
|
264
539
|
|
|
@@ -282,7 +557,7 @@ The GeminiLiveVoice class emits the following events:
|
|
|
282
557
|
|
|
283
558
|
**interrupt** (`event`): Emitted on barge-in when the user starts speaking over an in-flight model response. The server cancels any further audio for the current turn. Callback receives { type: 'user', timestamp: number }.
|
|
284
559
|
|
|
285
|
-
|
|
560
|
+
### Native-audio behavior
|
|
286
561
|
|
|
287
562
|
Native-audio Gemini Live models (any model whose ID contains `native-audio`, such as `gemini-2.5-flash-native-audio-preview-12-2025`) split text output across two channels:
|
|
288
563
|
|
|
@@ -293,7 +568,7 @@ On non-native-audio models there is no `output_audio_transcription` channel, so
|
|
|
293
568
|
|
|
294
569
|
Input transcription, output transcription, and barge-in detection (`realtime_input_config.activity_handling = 'START_OF_ACTIVITY_INTERRUPTS'`) are enabled automatically in the setup payload. You don't need extra configuration.
|
|
295
570
|
|
|
296
|
-
|
|
571
|
+
### Available models
|
|
297
572
|
|
|
298
573
|
The following Gemini Live models are available:
|
|
299
574
|
|
|
@@ -305,7 +580,7 @@ The following Gemini Live models are available:
|
|
|
305
580
|
- `gemini-live-2.5-flash-preview`
|
|
306
581
|
- `gemini-2.6.flash-preview-tts`
|
|
307
582
|
|
|
308
|
-
|
|
583
|
+
### Available voices
|
|
309
584
|
|
|
310
585
|
The following voice options are available:
|
|
311
586
|
|
|
@@ -314,9 +589,9 @@ The following voice options are available:
|
|
|
314
589
|
- `Kore`: Neutral, professional
|
|
315
590
|
- `Fenrir`: Warm, approachable
|
|
316
591
|
|
|
317
|
-
|
|
592
|
+
### Authentication methods
|
|
318
593
|
|
|
319
|
-
|
|
594
|
+
#### Gemini API (Development)
|
|
320
595
|
|
|
321
596
|
The simplest method using an API key from [Google AI Studio](https://makersuite.google.com/app/apikey):
|
|
322
597
|
|
|
@@ -327,7 +602,7 @@ const voice = new GeminiLiveVoice({
|
|
|
327
602
|
})
|
|
328
603
|
```
|
|
329
604
|
|
|
330
|
-
|
|
605
|
+
#### Vertex AI (Production)
|
|
331
606
|
|
|
332
607
|
For production use with OAuth authentication and Google Cloud Platform:
|
|
333
608
|
|
|
@@ -356,9 +631,9 @@ const voice = new GeminiLiveVoice({
|
|
|
356
631
|
})
|
|
357
632
|
```
|
|
358
633
|
|
|
359
|
-
|
|
634
|
+
### Advanced features
|
|
360
635
|
|
|
361
|
-
|
|
636
|
+
#### Session Management
|
|
362
637
|
|
|
363
638
|
The Gemini Live API supports session resumption for handling network interruptions:
|
|
364
639
|
|
|
@@ -377,7 +652,7 @@ const voice = new GeminiLiveVoice({
|
|
|
377
652
|
})
|
|
378
653
|
```
|
|
379
654
|
|
|
380
|
-
|
|
655
|
+
#### Tool Calling
|
|
381
656
|
|
|
382
657
|
Enable the model to call functions during conversations:
|
|
383
658
|
|
|
@@ -402,7 +677,7 @@ voice.on('toolCall', ({ name, args, id }) => {
|
|
|
402
677
|
})
|
|
403
678
|
```
|
|
404
679
|
|
|
405
|
-
|
|
680
|
+
### Notes
|
|
406
681
|
|
|
407
682
|
- The Gemini Live API uses WebSockets for real-time communication
|
|
408
683
|
- Audio is processed as 16kHz PCM16 for input and 24kHz PCM16 for output
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@mastra/voice-google-gemini-live",
|
|
3
|
-
"version": "0.14.6
|
|
3
|
+
"version": "0.14.6",
|
|
4
4
|
"description": "Mastra Google Gemini Live API integration",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"files": [
|
|
@@ -27,7 +27,7 @@
|
|
|
27
27
|
"@google/genai": "^1.52.0",
|
|
28
28
|
"google-auth-library": "^10.9.1",
|
|
29
29
|
"ws": "^8.21.0",
|
|
30
|
-
"@mastra/schema-compat": "1.3.6
|
|
30
|
+
"@mastra/schema-compat": "1.3.6"
|
|
31
31
|
},
|
|
32
32
|
"devDependencies": {
|
|
33
33
|
"@types/node": "22.20.1",
|
|
@@ -40,10 +40,10 @@
|
|
|
40
40
|
"typescript": "^6.0.3",
|
|
41
41
|
"vitest": "4.1.10",
|
|
42
42
|
"zod": "^4.4.3",
|
|
43
|
-
"@internal/
|
|
44
|
-
"@internal/
|
|
45
|
-
"@internal/
|
|
46
|
-
"@internal/
|
|
43
|
+
"@internal/test-utils": "0.0.58",
|
|
44
|
+
"@internal/voice": "0.0.20",
|
|
45
|
+
"@internal/types-builder": "0.0.97",
|
|
46
|
+
"@internal/lint": "0.0.122"
|
|
47
47
|
},
|
|
48
48
|
"homepage": "https://mastra.ai",
|
|
49
49
|
"repository": {
|