@mastra/voice-google 0.14.0 → 0.14.1-alpha.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -18,7 +18,7 @@ const voiceAgent = new Agent({
18
18
  id: 'voice-agent',
19
19
  name: 'Voice Agent',
20
20
  instructions: 'You are a voice assistant that can help users with their tasks.',
21
- model: 'openai/gpt-5.5',
21
+ model: 'openai/gpt-5.6-sol',
22
22
  voice: new OpenAIVoice(),
23
23
  })
24
24
  ```
@@ -29,7 +29,7 @@ You can then use the following voice capabilities:
29
29
 
30
30
  Turn your agent's responses into natural-sounding speech using Mastra's TTS capabilities. Choose from multiple providers like OpenAI, ElevenLabs, and more.
31
31
 
32
- For detailed configuration options and advanced features, check out our [Text-to-Speech guide](https://mastra.ai/docs/voice/text-to-speech).
32
+ For detailed configuration options and advanced features, check out our [Text-to-Speech guide](https://mastra.ai/guides/voice/text-to-speech).
33
33
 
34
34
  **OpenAI**:
35
35
 
@@ -42,7 +42,7 @@ const voiceAgent = new Agent({
42
42
  id: 'voice-agent',
43
43
  name: 'Voice Agent',
44
44
  instructions: 'You are a voice assistant that can help users with their tasks.',
45
- model: 'openai/gpt-5.5',
45
+ model: 'openai/gpt-5.6-sol',
46
46
  voice: new OpenAIVoice(),
47
47
  })
48
48
 
@@ -70,7 +70,7 @@ const voiceAgent = new Agent({
70
70
  id: 'voice-agent',
71
71
  name: 'Voice Agent',
72
72
  instructions: 'You are a voice assistant that can help users with their tasks.',
73
- model: 'openai/gpt-5.5',
73
+ model: 'openai/gpt-5.6-sol',
74
74
  voice: new AzureVoice(),
75
75
  })
76
76
 
@@ -97,7 +97,7 @@ const voiceAgent = new Agent({
97
97
  id: 'voice-agent',
98
98
  name: 'Voice Agent',
99
99
  instructions: 'You are a voice assistant that can help users with their tasks.',
100
- model: 'openai/gpt-5.5',
100
+ model: 'openai/gpt-5.6-sol',
101
101
  voice: new ElevenLabsVoice(),
102
102
  })
103
103
 
@@ -124,7 +124,7 @@ const voiceAgent = new Agent({
124
124
  id: 'voice-agent',
125
125
  name: 'Voice Agent',
126
126
  instructions: 'You are a voice assistant that can help users with their tasks.',
127
- model: 'openai/gpt-5.5',
127
+ model: 'openai/gpt-5.6-sol',
128
128
  voice: new PlayAIVoice(),
129
129
  })
130
130
 
@@ -151,7 +151,7 @@ const voiceAgent = new Agent({
151
151
  id: 'voice-agent',
152
152
  name: 'Voice Agent',
153
153
  instructions: 'You are a voice assistant that can help users with their tasks.',
154
- model: 'openai/gpt-5.5',
154
+ model: 'openai/gpt-5.6-sol',
155
155
  voice: new GoogleVoice(),
156
156
  })
157
157
 
@@ -178,7 +178,7 @@ const voiceAgent = new Agent({
178
178
  id: 'voice-agent',
179
179
  name: 'Voice Agent',
180
180
  instructions: 'You are a voice assistant that can help users with their tasks.',
181
- model: 'openai/gpt-5.5',
181
+ model: 'openai/gpt-5.6-sol',
182
182
  voice: new CloudflareVoice(),
183
183
  })
184
184
 
@@ -205,7 +205,7 @@ const voiceAgent = new Agent({
205
205
  id: 'voice-agent',
206
206
  name: 'Voice Agent',
207
207
  instructions: 'You are a voice assistant that can help users with their tasks.',
208
- model: 'openai/gpt-5.5',
208
+ model: 'openai/gpt-5.6-sol',
209
209
  voice: new DeepgramVoice(),
210
210
  })
211
211
 
@@ -232,7 +232,7 @@ const voiceAgent = new Agent({
232
232
  id: 'voice-agent',
233
233
  name: 'Voice Agent',
234
234
  instructions: 'You are a voice assistant that can help users with their tasks.',
235
- model: 'openai/gpt-5.5',
235
+ model: 'openai/gpt-5.6-sol',
236
236
  voice: new InworldVoice(),
237
237
  })
238
238
 
@@ -259,7 +259,7 @@ const voiceAgent = new Agent({
259
259
  id: 'voice-agent',
260
260
  name: 'Voice Agent',
261
261
  instructions: 'You are a voice assistant that can help users with their tasks.',
262
- model: 'openai/gpt-5.5',
262
+ model: 'openai/gpt-5.6-sol',
263
263
  voice: new SpeechifyVoice(),
264
264
  })
265
265
 
@@ -286,7 +286,7 @@ const voiceAgent = new Agent({
286
286
  id: 'voice-agent',
287
287
  name: 'Voice Agent',
288
288
  instructions: 'You are a voice assistant that can help users with their tasks.',
289
- model: 'openai/gpt-5.5',
289
+ model: 'openai/gpt-5.6-sol',
290
290
  voice: new SarvamVoice(),
291
291
  })
292
292
 
@@ -313,7 +313,7 @@ const voiceAgent = new Agent({
313
313
  id: 'voice-agent',
314
314
  name: 'Voice Agent',
315
315
  instructions: 'You are a voice assistant that can help users with their tasks.',
316
- model: 'openai/gpt-5.5',
316
+ model: 'openai/gpt-5.6-sol',
317
317
  voice: new MurfVoice(),
318
318
  })
319
319
 
@@ -331,7 +331,7 @@ Visit the [Murf Voice Reference](https://mastra.ai/reference/voice/murf) for mor
331
331
 
332
332
  ### Speech to Text (STT)
333
333
 
334
- Transcribe spoken content using various providers like OpenAI, ElevenLabs, and more. For detailed configuration options and more, check out [Speech to Text](https://mastra.ai/docs/voice/speech-to-text).
334
+ Transcribe spoken content using providers like OpenAI, ElevenLabs, and more. For detailed configuration options and more, check out [Speech to Text](https://mastra.ai/guides/voice/speech-to-text).
335
335
 
336
336
  You can download a sample audio file from [here](https://github.com/mastra-ai/realtime-voice-demo/raw/refs/heads/main/how_can_i_help_you.mp3).
337
337
 
@@ -348,7 +348,7 @@ const voiceAgent = new Agent({
348
348
  id: 'voice-agent',
349
349
  name: 'Voice Agent',
350
350
  instructions: 'You are a voice assistant that can help users with their tasks.',
351
- model: 'openai/gpt-5.5',
351
+ model: 'openai/gpt-5.6-sol',
352
352
  voice: new OpenAIVoice(),
353
353
  })
354
354
 
@@ -377,7 +377,7 @@ const voiceAgent = new Agent({
377
377
  id: 'voice-agent',
378
378
  name: 'Voice Agent',
379
379
  instructions: 'You are a voice assistant that can help users with their tasks.',
380
- model: 'openai/gpt-5.5',
380
+ model: 'openai/gpt-5.6-sol',
381
381
  voice: new AzureVoice(),
382
382
  })
383
383
 
@@ -405,7 +405,7 @@ const voiceAgent = new Agent({
405
405
  id: 'voice-agent',
406
406
  name: 'Voice Agent',
407
407
  instructions: 'You are a voice assistant that can help users with their tasks.',
408
- model: 'openai/gpt-5.5',
408
+ model: 'openai/gpt-5.6-sol',
409
409
  voice: new ElevenLabsVoice(),
410
410
  })
411
411
 
@@ -433,7 +433,7 @@ const voiceAgent = new Agent({
433
433
  id: 'voice-agent',
434
434
  name: 'Voice Agent',
435
435
  instructions: 'You are a voice assistant that can help users with their tasks.',
436
- model: 'openai/gpt-5.5',
436
+ model: 'openai/gpt-5.6-sol',
437
437
  voice: new GoogleVoice(),
438
438
  })
439
439
 
@@ -461,7 +461,7 @@ const voiceAgent = new Agent({
461
461
  id: 'voice-agent',
462
462
  name: 'Voice Agent',
463
463
  instructions: 'You are a voice assistant that can help users with their tasks.',
464
- model: 'openai/gpt-5.5',
464
+ model: 'openai/gpt-5.6-sol',
465
465
  voice: new CloudflareVoice(),
466
466
  })
467
467
 
@@ -489,7 +489,7 @@ const voiceAgent = new Agent({
489
489
  id: 'voice-agent',
490
490
  name: 'Voice Agent',
491
491
  instructions: 'You are a voice assistant that can help users with their tasks.',
492
- model: 'openai/gpt-5.5',
492
+ model: 'openai/gpt-5.6-sol',
493
493
  voice: new DeepgramVoice(),
494
494
  })
495
495
 
@@ -517,7 +517,7 @@ const voiceAgent = new Agent({
517
517
  id: 'voice-agent',
518
518
  name: 'Voice Agent',
519
519
  instructions: 'You are a voice assistant that can help users with their tasks.',
520
- model: 'openai/gpt-5.5',
520
+ model: 'openai/gpt-5.6-sol',
521
521
  voice: new InworldVoice(),
522
522
  })
523
523
 
@@ -545,7 +545,7 @@ const voiceAgent = new Agent({
545
545
  id: 'voice-agent',
546
546
  name: 'Voice Agent',
547
547
  instructions: 'You are a voice assistant that can help users with their tasks.',
548
- model: 'openai/gpt-5.5',
548
+ model: 'openai/gpt-5.6-sol',
549
549
  voice: new SarvamVoice(),
550
550
  })
551
551
 
@@ -564,7 +564,7 @@ Visit the [Sarvam Voice Reference](https://mastra.ai/reference/voice/sarvam) for
564
564
 
565
565
  ### Speech to Speech (STS)
566
566
 
567
- Create conversational experiences with speech-to-speech capabilities. The unified API enables real-time voice interactions between users and AI agents. For detailed configuration options and advanced features, check out [Speech to Speech](https://mastra.ai/docs/voice/speech-to-speech).
567
+ Create conversational experiences with speech-to-speech capabilities. The unified API enables real-time voice interactions between users and AI agents. For detailed configuration options and advanced features, check out [Speech to Speech](https://mastra.ai/guides/voice/speech-to-speech).
568
568
 
569
569
  **OpenAI**:
570
570
 
@@ -577,7 +577,7 @@ const voiceAgent = new Agent({
577
577
  id: 'voice-agent',
578
578
  name: 'Voice Agent',
579
579
  instructions: 'You are a voice assistant that can help users with their tasks.',
580
- model: 'openai/gpt-5.5',
580
+ model: 'openai/gpt-5.6-sol',
581
581
  voice: new OpenAIRealtimeVoice(),
582
582
  })
583
583
 
@@ -607,7 +607,7 @@ const voiceAgent = new Agent({
607
607
  id: 'voice-agent',
608
608
  name: 'Voice Agent',
609
609
  instructions: 'You are a voice assistant that can help users with their tasks.',
610
- model: 'openai/gpt-5.5',
610
+ model: 'openai/gpt-5.6-sol',
611
611
  voice: new GeminiLiveVoice({
612
612
  // Live API mode
613
613
  apiKey: process.env.GOOGLE_API_KEY,
@@ -656,7 +656,7 @@ const voiceAgent = new Agent({
656
656
  id: 'voice-agent',
657
657
  name: 'Voice Agent',
658
658
  instructions: 'You are a voice assistant that can help users with their tasks.',
659
- model: 'openai/gpt-5.5',
659
+ model: 'openai/gpt-5.6-sol',
660
660
  voice: new NovaSonicVoice({
661
661
  region: 'us-east-1',
662
662
  speaker: 'matthew',
@@ -699,7 +699,7 @@ const voiceAgent = new Agent({
699
699
  id: 'voice-agent',
700
700
  name: 'Voice Agent',
701
701
  instructions: 'You are a voice assistant that can help users with their tasks.',
702
- model: 'openai/gpt-5.5',
702
+ model: 'openai/gpt-5.6-sol',
703
703
  voice: new InworldRealtimeVoice({
704
704
  apiKey: process.env.INWORLD_API_KEY,
705
705
  model: 'inworld/models/gemma-4-26b-a4b-it',
@@ -773,6 +773,10 @@ await voiceAgent.voice.send(micStream)
773
773
 
774
774
  Visit the [xAI Realtime Voice Reference](https://mastra.ai/reference/voice/xai-realtime) for more information on the xAI voice provider.
775
775
 
776
+ ### Realtime voice
777
+
778
+ Run live calls a user can talk over, in the browser or over the phone. Mastra hands the audio loop to LiveKit, which covers voice activity detection, semantic turn detection, and barge-in, while your agent generates each reply with its own model, tools, and memory. For setup and configuration options, check out [Realtime voice](https://mastra.ai/guides/voice/realtime-voice).
779
+
776
780
  ## Voice configuration
777
781
 
778
782
  Each voice provider can be configured with different models and options. Below are the detailed configuration options for all supported providers:
@@ -1134,7 +1138,7 @@ const voiceAgent = new Agent({
1134
1138
  id: 'aisdk-voice-agent',
1135
1139
  name: 'AI SDK Voice Agent',
1136
1140
  instructions: 'You are a helpful assistant with voice capabilities.',
1137
- model: 'openai/gpt-5.5',
1141
+ model: 'openai/gpt-5.6-sol',
1138
1142
  voice,
1139
1143
  })
1140
1144
  ```
@@ -1248,5 +1252,6 @@ For more information on the CompositeVoice, refer to the [CompositeVoice Referen
1248
1252
  - [AWS Nova Sonic Voice](https://mastra.ai/reference/voice/aws-nova-sonic)
1249
1253
  - [Deepgram Voice](https://mastra.ai/reference/voice/deepgram)
1250
1254
  - [Inworld Voice](https://mastra.ai/reference/voice/inworld)
1255
+ - [LiveKit](https://mastra.ai/reference/voice/livekit)
1251
1256
  - [PlayAI Voice](https://mastra.ai/reference/voice/playai)
1252
1257
  - [Voice Examples](https://github.com/mastra-ai/voice-examples)
@@ -109,7 +109,17 @@ Converts speech to text using Google Cloud Speech-to-Text service. Supports both
109
109
 
110
110
  Pass `v2: true` to use the Cloud Speech-to-Text v2 API, which supports additional audio formats like AAC-in-MP4 (iOS Safari).
111
111
 
112
+ The v2 `recognize` call is IAM-authorized and does not accept API-key-only authentication. Configure service account credentials on the `listeningModel` (or set `GOOGLE_APPLICATION_CREDENTIALS`) and set `GOOGLE_CLOUD_PROJECT` so the recognizer path can be resolved, even when `vertexAI` is not enabled.
113
+
112
114
  ```typescript
115
+ import { GoogleVoice } from '@mastra/voice-google'
116
+
117
+ // v2 listen() requires service account credentials, not just GOOGLE_API_KEY.
118
+ // Set GOOGLE_CLOUD_PROJECT so the recognizer path can be resolved.
119
+ const voice = new GoogleVoice({
120
+ listeningModel: { keyFilename: process.env.GOOGLE_APPLICATION_CREDENTIALS },
121
+ })
122
+
113
123
  const transcript = await voice.listen(iosSafariAacStream, {
114
124
  v2: true,
115
125
  config: {
@@ -118,6 +128,8 @@ const transcript = await voice.listen(iosSafariAacStream, {
118
128
  })
119
129
  ```
120
130
 
131
+ > **Note:** `listen({ v2: true })` fails with `PERMISSION_DENIED` on `speech.recognizers.recognize` when only `GOOGLE_API_KEY` is set. An API-key request carries no OAuth identity, so granting `roles/speech.client` to a user account does not help — the role must be granted to the service account presented in the request. This applies regardless of the `vertexAI` setting; `speak()` and v1 `listen()` still work with an API key alone.
132
+
121
133
  **audioStream** (`NodeJS.ReadableStream`): Audio stream to transcribe
122
134
 
123
135
  **options** (`GoogleListenOptionsV2`): v2 recognition options
@@ -162,7 +174,7 @@ The Google Voice provider supports two authentication methods:
162
174
 
163
175
  ### Standard Mode (API Key)
164
176
 
165
- Uses a Google Cloud API key for authentication. Suitable for development and basic use cases.
177
+ Uses a Google Cloud API key for authentication. Covers `speak()` and v1 `listen()`. It does not cover `listen({ v2: true })`, which is IAM-authorized and requires service account credentials (see [v2](#v2)).
166
178
 
167
179
  ```typescript
168
180
  // Using environment variable (GOOGLE_API_KEY)
@@ -238,6 +250,8 @@ For Speech-to-Text:
238
250
 
239
251
  - `roles/speech.client` - Speech-to-Text Client
240
252
 
253
+ Grant `roles/speech.client` to the service account whose credentials the request presents (via `keyFilename`, `credentials`, or `GOOGLE_APPLICATION_CREDENTIALS`). This role is required for `listen({ v2: true })` specifically, not only for Vertex AI mode. Granting it to a user account has no effect on API-key-only requests, which carry no identity to authorize.
254
+
241
255
  #### OAuth Scopes
242
256
 
243
257
  For synchronous Text-to-Speech synthesis:
@@ -269,6 +283,8 @@ For long-audio Text-to-Speech operations:
269
283
 
270
284
  6. The `listen()` method supports various recognition configurations through the Google Cloud Speech-to-Text API.
271
285
 
272
- 7. Available voices can be filtered by language code using the `getSpeakers()` method.
286
+ 7. `listen({ v2: true })` requires service account credentials and `GOOGLE_CLOUD_PROJECT`; it fails with `PERMISSION_DENIED` when only `GOOGLE_API_KEY` is set. `speak()` and v1 `listen()` work with an API key alone.
287
+
288
+ 8. Available voices can be filtered by language code using the `getSpeakers()` method.
273
289
 
274
- 8. Vertex AI mode provides enterprise features including IAM control, audit logs, and project-level billing.
290
+ 9. Vertex AI mode provides enterprise features including IAM control, audit logs, and project-level billing.