@mastra/voice-google-gemini-live 0.14.3 → 0.14.4-alpha.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +7 -0
- package/dist/docs/SKILL.md +1 -1
- package/dist/docs/assets/SOURCE_MAP.json +1 -1
- package/dist/docs/references/docs-voice-overview.md +3 -3
- package/dist/docs/references/docs-voice-speech-to-speech.md +81 -1
- package/dist/docs/references/reference-voice-google-gemini-live.md +3 -3
- package/package.json +9 -9
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,12 @@
|
|
|
1
1
|
# @mastra/voice-google-gemini-live
|
|
2
2
|
|
|
3
|
+
## 0.14.4-alpha.0
|
|
4
|
+
|
|
5
|
+
### Patch Changes
|
|
6
|
+
|
|
7
|
+
- Updated dependencies [[`6789ab4`](https://github.com/mastra-ai/mastra/commit/6789ab4191ddcd32a932898b360b191e80cee1a9)]:
|
|
8
|
+
- @mastra/schema-compat@1.3.4-alpha.0
|
|
9
|
+
|
|
3
10
|
## 0.14.3
|
|
4
11
|
|
|
5
12
|
### Patch Changes
|
package/dist/docs/SKILL.md
CHANGED
|
@@ -3,7 +3,7 @@ name: mastra-voice-google-gemini-live
|
|
|
3
3
|
description: Documentation for @mastra/voice-google-gemini-live. Use when working with @mastra/voice-google-gemini-live APIs, configuration, or implementation.
|
|
4
4
|
metadata:
|
|
5
5
|
package: "@mastra/voice-google-gemini-live"
|
|
6
|
-
version: "0.14.
|
|
6
|
+
version: "0.14.4-alpha.0"
|
|
7
7
|
---
|
|
8
8
|
|
|
9
9
|
## When to use
|
|
@@ -4,9 +4,9 @@
|
|
|
4
4
|
|
|
5
5
|
Mastra's Voice system provides a unified interface for voice interactions, enabling text-to-speech (TTS), speech-to-text (STT), and real-time speech-to-speech (STS) capabilities in your applications.
|
|
6
6
|
|
|
7
|
-
##
|
|
7
|
+
## Add voice to agents
|
|
8
8
|
|
|
9
|
-
|
|
9
|
+
Pass a voice provider to an agent with the `voice` property. The same property supports text-to-speech (TTS), speech-to-text (STT), and real-time speech-to-speech (STS), depending on the provider you configure.
|
|
10
10
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Agent } from '@mastra/core/agent'
|
|
@@ -1139,7 +1139,7 @@ const voiceAgent = new Agent({
|
|
|
1139
1139
|
})
|
|
1140
1140
|
```
|
|
1141
1141
|
|
|
1142
|
-
### Using
|
|
1142
|
+
### Using multiple voice providers
|
|
1143
1143
|
|
|
1144
1144
|
This example demonstrates how to create and use two different voice providers in Mastra: OpenAI for speech-to-text (STT) and PlayAI for text-to-speech (TTS).
|
|
1145
1145
|
|
|
@@ -54,7 +54,87 @@ const micStream = getMicrophoneStream()
|
|
|
54
54
|
await agent.voice.send(micStream)
|
|
55
55
|
```
|
|
56
56
|
|
|
57
|
-
For
|
|
57
|
+
For a broader overview of voice providers on agents, see [Voice in Mastra](https://mastra.ai/docs/voice/overview).
|
|
58
|
+
|
|
59
|
+
## Use tools in realtime sessions
|
|
60
|
+
|
|
61
|
+
Realtime voice providers can use tools configured on the agent. Add the tools to the `Agent` definition, then connect and send audio through the voice provider:
|
|
62
|
+
|
|
63
|
+
```typescript
|
|
64
|
+
import { Agent } from '@mastra/core/agent'
|
|
65
|
+
import { OpenAIRealtimeVoice } from '@mastra/voice-openai-realtime'
|
|
66
|
+
import { calculate, search } from '../tools'
|
|
67
|
+
|
|
68
|
+
export const agent = new Agent({
|
|
69
|
+
id: 'speech-to-speech-agent',
|
|
70
|
+
name: 'Speech-to-Speech Agent',
|
|
71
|
+
instructions: 'You are a helpful assistant with speech-to-speech capabilities.',
|
|
72
|
+
model: 'openai/gpt-5.5',
|
|
73
|
+
tools: {
|
|
74
|
+
search,
|
|
75
|
+
calculate,
|
|
76
|
+
},
|
|
77
|
+
voice: new OpenAIRealtimeVoice(),
|
|
78
|
+
})
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
## Listen for realtime events
|
|
82
|
+
|
|
83
|
+
Realtime voice providers emit events you can use to update your UI, play assistant audio, log transcriptions, and handle errors:
|
|
84
|
+
|
|
85
|
+
```typescript
|
|
86
|
+
agent.voice.on('speaking', ({ audio }) => {
|
|
87
|
+
playAudio(audio)
|
|
88
|
+
})
|
|
89
|
+
|
|
90
|
+
agent.voice.on('writing', ({ text, role }) => {
|
|
91
|
+
console.log(`${role}: ${text}`)
|
|
92
|
+
})
|
|
93
|
+
|
|
94
|
+
agent.voice.on('error', error => {
|
|
95
|
+
console.error('Voice error:', error)
|
|
96
|
+
})
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
Event names and payloads vary by provider. Check the provider section below or the provider reference for the full event list.
|
|
100
|
+
|
|
101
|
+
## Per-session voice instances
|
|
102
|
+
|
|
103
|
+
A static `voice` instance is shared across every request. This works for one-shot text-to-speech, but real-time and speech-to-speech providers store session state such as the WebSocket connection, tools, instructions, and request context. If one agent handles several live sessions at once, a shared instance can let one session overwrite another session's state.
|
|
104
|
+
|
|
105
|
+
Provide `voice` as a resolver when each live session needs its own voice instance. Mastra runs the resolver on each `getVoice()` call and returns a fresh instance for that request context:
|
|
106
|
+
|
|
107
|
+
```typescript
|
|
108
|
+
import { Agent } from '@mastra/core/agent'
|
|
109
|
+
import { RequestContext } from '@mastra/core/request-context'
|
|
110
|
+
import { OpenAIRealtimeVoice } from '@mastra/voice-openai-realtime'
|
|
111
|
+
|
|
112
|
+
export const agent = new Agent({
|
|
113
|
+
id: 'support-line',
|
|
114
|
+
name: 'Support Line',
|
|
115
|
+
instructions: ({ requestContext }) => `Help user ${requestContext.get('user')}.`,
|
|
116
|
+
model: 'openai/gpt-5.5',
|
|
117
|
+
voice: ({ requestContext }) =>
|
|
118
|
+
new OpenAIRealtimeVoice({
|
|
119
|
+
apiKey: requestContext.get('apiKey'),
|
|
120
|
+
}),
|
|
121
|
+
})
|
|
122
|
+
|
|
123
|
+
const requestContext = new RequestContext()
|
|
124
|
+
requestContext.set('user', 'user-123')
|
|
125
|
+
requestContext.set('apiKey', process.env.OPENAI_API_KEY)
|
|
126
|
+
|
|
127
|
+
const voice = await agent.getVoice({ requestContext })
|
|
128
|
+
await voice.connect()
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
When you use a resolver:
|
|
132
|
+
|
|
133
|
+
- Each call to `getVoice()` returns a new instance, so concurrent sessions don't share state.
|
|
134
|
+
- Mastra doesn't add tools or instructions to a resolver instance. Configure them inside the resolver or on the provider.
|
|
135
|
+
- You own the returned instance lifecycle, so call `disconnect()` or `close()` when the session ends.
|
|
136
|
+
|
|
137
|
+
The `agent.voice` getter has no request context, so it throws when `voice` is a resolver. Use `agent.getVoice({ requestContext })` instead.
|
|
58
138
|
|
|
59
139
|
## Google Gemini Live (Realtime)
|
|
60
140
|
|
|
@@ -162,7 +162,7 @@ await voice.sendContext([
|
|
|
162
162
|
await voice.send(micStream)
|
|
163
163
|
```
|
|
164
164
|
|
|
165
|
-
**turns** (`IncrementalTurn[]`): Prior conversation turns to seed into the session. Each turn has a
|
|
165
|
+
**turns** (`IncrementalTurn[]`): Prior conversation turns to seed into the session. Each turn has a role ("user" or "assistant") and content string. Both roles are supported on newer models (e.g. gemini-2.5-flash-native-audio-preview-12-2025). Some older models only accept user-role turns.
|
|
166
166
|
|
|
167
167
|
**options** (`object`): Optional configuration.
|
|
168
168
|
|
|
@@ -266,9 +266,9 @@ The GeminiLiveVoice class emits the following events:
|
|
|
266
266
|
|
|
267
267
|
**speaking** (`event`): Emitted with audio metadata. Callback receives { audioData?: Int16Array, sampleRate?: number }.
|
|
268
268
|
|
|
269
|
-
**writing** (`event`): Emitted when transcribed text is available. Callback receives { text: string, role: 'assistant' | 'user' }. On native-audio models the assistant transcript is driven by the server's
|
|
269
|
+
**writing** (`event`): Emitted when transcribed text is available. Callback receives { text: string, role: 'assistant' | 'user' }. On native-audio models the assistant transcript is driven by the server's output\_audio\_transcription channel rather than modelTurn.parts.text.
|
|
270
270
|
|
|
271
|
-
**thinking** (`event`): Emitted on native-audio models with the model's chain-of-thought / reasoning text from
|
|
271
|
+
**thinking** (`event`): Emitted on native-audio models with the model's chain-of-thought / reasoning text from modelTurn.parts.text. Callback receives { text: string }. Does not fire on non-native-audio models, where modelTurn.parts.text is the spoken response and is emitted as writing instead.
|
|
272
272
|
|
|
273
273
|
**session** (`event`): Emitted on session state changes. Callback receives { state: 'connecting' | 'connected' | 'disconnected' | 'disconnecting' | 'updated', config?: object }.
|
|
274
274
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@mastra/voice-google-gemini-live",
|
|
3
|
-
"version": "0.14.
|
|
3
|
+
"version": "0.14.4-alpha.0",
|
|
4
4
|
"description": "Mastra Google Gemini Live API integration",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"files": [
|
|
@@ -27,23 +27,23 @@
|
|
|
27
27
|
"@google/genai": "^1.52.0",
|
|
28
28
|
"google-auth-library": "^10.6.2",
|
|
29
29
|
"ws": "^8.21.0",
|
|
30
|
-
"@mastra/schema-compat": "1.3.
|
|
30
|
+
"@mastra/schema-compat": "1.3.4-alpha.0"
|
|
31
31
|
},
|
|
32
32
|
"devDependencies": {
|
|
33
33
|
"@types/node": "22.19.21",
|
|
34
34
|
"@types/ws": "^8.18.1",
|
|
35
|
-
"@vitest/coverage-v8": "4.1.
|
|
36
|
-
"@vitest/ui": "4.1.
|
|
35
|
+
"@vitest/coverage-v8": "4.1.9",
|
|
36
|
+
"@vitest/ui": "4.1.9",
|
|
37
37
|
"eslint": "^10.4.1",
|
|
38
38
|
"tsup": "^8.5.1",
|
|
39
39
|
"tsx": "^4.22.4",
|
|
40
40
|
"typescript": "^6.0.3",
|
|
41
|
-
"vitest": "4.1.
|
|
41
|
+
"vitest": "4.1.9",
|
|
42
42
|
"zod": "^4.4.3",
|
|
43
|
-
"@internal/lint": "0.0.
|
|
44
|
-
"@internal/
|
|
45
|
-
"@internal/
|
|
46
|
-
"@internal/
|
|
43
|
+
"@internal/lint": "0.0.113",
|
|
44
|
+
"@internal/test-utils": "0.0.49",
|
|
45
|
+
"@internal/types-builder": "0.0.88",
|
|
46
|
+
"@internal/voice": "0.0.13"
|
|
47
47
|
},
|
|
48
48
|
"homepage": "https://mastra.ai",
|
|
49
49
|
"repository": {
|