@mastra/voice-google-gemini-live 0.13.0 → 0.14.0-alpha.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,33 @@
1
1
  # @mastra/voice-google-gemini-live
2
2
 
3
+ ## 0.14.0-alpha.0
4
+
5
+ ### Minor Changes
6
+
7
+ - Added `sendContext()` method to `GeminiLiveVoice` for seeding conversation history into a fresh voice session without triggering a model response. This lets apps replay prior turns from Mastra Memory (or any external store) on a cold connect so the model has full context before the user speaks — enabling seamless handoff between text chat and voice on a shared thread. ([#18286](https://github.com/mastra-ai/mastra/pull/18286))
8
+
9
+ **Usage:**
10
+
11
+ ```ts
12
+ await voice.connect();
13
+
14
+ await voice.sendContext([
15
+ { role: 'user', content: 'What is the weather?' },
16
+ { role: 'assistant', content: 'It is 72°F in San Francisco.' },
17
+ ]);
18
+
19
+ // Model stays silent until the user actually speaks
20
+ await voice.send(micStream);
21
+ ```
22
+
23
+ ### Patch Changes
24
+
25
+ - Fix resumeSession() always timing out. Session resumption now works end-to-end: new sessions request server-issued tokens, inbound handles are stored and emitted, and resuming reconnects with the correct handle in the setup frame. ([#18190](https://github.com/mastra-ai/mastra/pull/18190))
26
+
27
+ - Fixed realtime audio streaming being immediately rejected by the Gemini Live API. Audio frames now use the current API format, replacing a deprecated payload shape that caused the connection to close on the first frame. ([#18291](https://github.com/mastra-ai/mastra/pull/18291))
28
+
29
+ The `session` event for disconnections now includes `code` and `reason` fields, so consumers can see why the server closed the connection.
30
+
3
31
  ## 0.13.0
4
32
 
5
33
  ### Minor Changes
@@ -3,7 +3,7 @@ name: mastra-voice-google-gemini-live
3
3
  description: Documentation for @mastra/voice-google-gemini-live. Use when working with @mastra/voice-google-gemini-live APIs, configuration, or implementation.
4
4
  metadata:
5
5
  package: "@mastra/voice-google-gemini-live"
6
- version: "0.13.0"
6
+ version: "0.14.0-alpha.0"
7
7
  ---
8
8
 
9
9
  ## When to use
@@ -1,5 +1,5 @@
1
1
  {
2
- "version": "0.13.0",
2
+ "version": "0.14.0-alpha.0",
3
3
  "package": "@mastra/voice-google-gemini-live",
4
4
  "exports": {},
5
5
  "modules": {}
@@ -146,6 +146,28 @@ Converts text to speech and sends it to the model. Can accept either a string or
146
146
 
147
147
  Returns: `Promise<void>` (responses are emitted via `speaker` and `writing` events)
148
148
 
149
+ ### `sendContext()`
150
+
151
+ Sends conversation history into the live session without triggering a model response. Use this to seed prior turns (e.g. from Mastra Memory) on a cold connect so the model has context before the user speaks.
152
+
153
+ ```typescript
154
+ await voice.sendContext([
155
+ { role: 'user', content: 'What is the weather?' },
156
+ { role: 'assistant', content: 'It is 72°F in San Francisco.' },
157
+ ])
158
+
159
+ // Model stays silent until the user actually speaks.
160
+ await voice.send(micStream)
161
+ ```
162
+
163
+ **turns** (`IncrementalTurn[]`): Prior conversation turns to seed into the session. Each turn has a \`role\` ("user" or "assistant") and \`content\` string. Both roles are supported on newer models (e.g. \`gemini-2.5-flash-native-audio-preview-12-2025\`). Some older models only accept user-role turns.
164
+
165
+ **options** (`object`): Optional configuration.
166
+
167
+ **options.turnComplete** (`boolean`): Whether to mark the turn as complete and trigger a model response.
168
+
169
+ Returns: `Promise<void>`
170
+
149
171
  ### `listen()`
150
172
 
151
173
  Processes audio input for speech recognition. Takes a readable stream of audio data and returns the transcribed text.
package/dist/index.cjs CHANGED
@@ -580,12 +580,10 @@ var AudioStreamManager = class {
580
580
  } else {
581
581
  return {
582
582
  realtime_input: {
583
- media_chunks: [
584
- {
585
- mime_type: "audio/pcm",
586
- data: audioData
587
- }
588
- ]
583
+ audio: {
584
+ mime_type: "audio/pcm",
585
+ data: audioData
586
+ }
589
587
  }
590
588
  };
591
589
  }
@@ -1808,13 +1806,9 @@ var GeminiLiveVoice = class _GeminiLiveVoice extends MastraVoice {
1808
1806
  this.connectionManager.setWebSocket(this.ws);
1809
1807
  this.setupEventListeners();
1810
1808
  await this.connectionManager.waitForOpen();
1811
- if (this.isResuming && this.sessionHandle) {
1812
- await this.sendSessionResumption();
1813
- } else {
1814
- this.sendInitialConfig();
1815
- this.sessionStartTime = Date.now();
1816
- this.sessionId = crypto.randomUUID();
1817
- }
1809
+ this.sendInitialConfig();
1810
+ this.sessionStartTime = Date.now();
1811
+ this.sessionId = crypto.randomUUID();
1818
1812
  await this.waitForSessionCreated();
1819
1813
  this.state = "connected";
1820
1814
  this.emit("session", {
@@ -1852,10 +1846,6 @@ var GeminiLiveVoice = class _GeminiLiveVoice extends MastraVoice {
1852
1846
  clearTimeout(this.sessionDurationTimeout);
1853
1847
  this.sessionDurationTimeout = void 0;
1854
1848
  }
1855
- if (this.options.sessionConfig?.enableResumption && this.sessionId) {
1856
- this.sessionHandle = this.sessionId;
1857
- this.log("Session handle saved for resumption", { handle: this.sessionHandle });
1858
- }
1859
1849
  if (this.ws) {
1860
1850
  this.connectionManager.close();
1861
1851
  this.ws = void 0;
@@ -1930,6 +1920,56 @@ var GeminiLiveVoice = class _GeminiLiveVoice extends MastraVoice {
1930
1920
  throw this.createAndEmitError("audio_processing_error" /* AUDIO_PROCESSING_ERROR */, "Failed to send text message", error);
1931
1921
  }
1932
1922
  }
1923
+ /**
1924
+ * Send conversation history into the live session without triggering a model response.
1925
+ *
1926
+ * Maps to a single Gemini Live `client_content` frame with `turnComplete` defaulting
1927
+ * to `false`, which loads the turns into context silently. The model only responds
1928
+ * once a subsequent turn completes (e.g. via {@link speak} or user audio).
1929
+ *
1930
+ * @param turns Prior conversation turns to seed into the session.
1931
+ * @param options.turnComplete Whether to mark the turn as complete (default `false`).
1932
+ *
1933
+ * @example
1934
+ * ```typescript
1935
+ * await voice.connect();
1936
+ *
1937
+ * // Replay prior conversation without triggering a reply.
1938
+ * await voice.sendContext([
1939
+ * { role: 'user', content: 'What is the weather?' },
1940
+ * { role: 'assistant', content: 'It is 72°F in San Francisco.' },
1941
+ * ]);
1942
+ *
1943
+ * // Agent stays silent until the user actually speaks.
1944
+ * await voice.send(micStream);
1945
+ * ```
1946
+ */
1947
+ async sendContext(turns, options) {
1948
+ this.validateConnectionState();
1949
+ if (!turns || turns.length === 0) {
1950
+ this.log("sendContext called with empty turns, skipping");
1951
+ return;
1952
+ }
1953
+ const message = {
1954
+ client_content: {
1955
+ turns: turns.map((t) => ({
1956
+ role: t.role === "assistant" ? "model" : "user",
1957
+ parts: [{ text: t.content }]
1958
+ })),
1959
+ turnComplete: options?.turnComplete ?? false
1960
+ }
1961
+ };
1962
+ try {
1963
+ this.sendEvent("client_content", message);
1964
+ this.log("Context seeded", { turnCount: turns.length, turnComplete: options?.turnComplete ?? false });
1965
+ for (const turn of turns) {
1966
+ this.addToContext(turn.role, turn.content);
1967
+ }
1968
+ } catch (error) {
1969
+ this.log("Failed to send context", error);
1970
+ throw this.createAndEmitError("audio_processing_error" /* AUDIO_PROCESSING_ERROR */, "Failed to send context", error);
1971
+ }
1972
+ }
1933
1973
  /**
1934
1974
  * Send audio stream for processing
1935
1975
  */
@@ -2267,30 +2307,6 @@ var GeminiLiveVoice = class _GeminiLiveVoice extends MastraVoice {
2267
2307
  * Send session resumption message
2268
2308
  * @private
2269
2309
  */
2270
- async sendSessionResumption() {
2271
- if (!this.sessionHandle) {
2272
- throw new Error("No session handle available for resumption");
2273
- }
2274
- const context = this.contextManager.getContextArray();
2275
- const resumeMessage = {
2276
- session_resume: {
2277
- handle: this.sessionHandle,
2278
- ...context.length > 0 && {
2279
- context
2280
- }
2281
- }
2282
- };
2283
- try {
2284
- if (this.ws?.readyState !== ws.WebSocket.OPEN) {
2285
- throw new Error("WebSocket not ready for session resumption");
2286
- }
2287
- this.sendEvent("session_resume", resumeMessage);
2288
- this.log("Session resumption message sent", { handle: this.sessionHandle });
2289
- } catch (error) {
2290
- this.log("Failed to send session resumption", error);
2291
- throw new Error(`Failed to send session resumption: ${error instanceof Error ? error.message : "Unknown error"}`);
2292
- }
2293
- }
2294
2310
  /**
2295
2311
  * Start monitoring session duration
2296
2312
  * @private
@@ -2362,7 +2378,7 @@ var GeminiLiveVoice = class _GeminiLiveVoice extends MastraVoice {
2362
2378
  this.ws.on("close", (code, reason) => {
2363
2379
  this.log("WebSocket connection closed", { code, reason: reason.toString() });
2364
2380
  this.state = "disconnected";
2365
- this.emit("session", { state: "disconnected" });
2381
+ this.emit("session", { state: "disconnected", code, reason: reason.toString() });
2366
2382
  });
2367
2383
  this.ws.on("error", (error) => {
2368
2384
  this.log("WebSocket error", error);
@@ -2413,6 +2429,18 @@ var GeminiLiveVoice = class _GeminiLiveVoice extends MastraVoice {
2413
2429
  } else if (data.usageMetadata) {
2414
2430
  this.log("Processing usage metadata message");
2415
2431
  this.handleUsageUpdate(data);
2432
+ }
2433
+ if (data.sessionResumptionUpdate) {
2434
+ this.log("Processing session resumption update", data.sessionResumptionUpdate);
2435
+ if (data.sessionResumptionUpdate.resumable && data.sessionResumptionUpdate.newHandle) {
2436
+ this.sessionHandle = data.sessionResumptionUpdate.newHandle;
2437
+ this.log("Session handle updated from server", { handle: this.sessionHandle });
2438
+ this.emit("sessionHandle", {
2439
+ handle: this.sessionHandle,
2440
+ expiresAt: new Date(Date.now() + 2 * 60 * 60 * 1e3)
2441
+ // 2h TTL for AI Studio
2442
+ });
2443
+ }
2416
2444
  } else if (data.sessionEnd) {
2417
2445
  this.log("Processing session end message");
2418
2446
  this.handleSessionEnd(data);
@@ -2820,6 +2848,11 @@ var GeminiLiveVoice = class _GeminiLiveVoice extends MastraVoice {
2820
2848
  // event declared in `GeminiLiveEventMap`.
2821
2849
  realtime_input_config: {
2822
2850
  activity_handling: "START_OF_ACTIVITY_INTERRUPTS"
2851
+ },
2852
+ // Session resumption: empty object requests server-issued tokens on new sessions;
2853
+ // { handle } resumes a previous session. Only included when enableResumption is set.
2854
+ ...this.options.sessionConfig?.enableResumption && {
2855
+ session_resumption: this.isResuming && this.sessionHandle ? { handle: this.sessionHandle } : {}
2823
2856
  }
2824
2857
  }
2825
2858
  };