@memberjunction/ai-realtime-client 6.1.1 → 6.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/drivers/geminiRealtimeClient.d.ts +130 -22
- package/dist/drivers/geminiRealtimeClient.d.ts.map +1 -1
- package/dist/drivers/geminiRealtimeClient.js +558 -45
- package/dist/drivers/geminiRealtimeClient.js.map +1 -1
- package/dist/drivers/openAILiveClient.d.ts +11 -0
- package/dist/drivers/openAILiveClient.d.ts.map +1 -1
- package/dist/drivers/openAILiveClient.js +29 -4
- package/dist/drivers/openAILiveClient.js.map +1 -1
- package/dist/generic/baseRealtimeClient.d.ts +53 -1
- package/dist/generic/baseRealtimeClient.d.ts.map +1 -1
- package/dist/generic/baseRealtimeClient.js +61 -0
- package/dist/generic/baseRealtimeClient.js.map +1 -1
- package/dist/index.d.ts +2 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -0
- package/dist/index.js.map +1 -1
- package/dist/media/channelVideoSource.d.ts +83 -0
- package/dist/media/channelVideoSource.d.ts.map +1 -0
- package/dist/media/channelVideoSource.js +117 -0
- package/dist/media/channelVideoSource.js.map +1 -0
- package/dist/media/frameCapture.d.ts +73 -0
- package/dist/media/frameCapture.d.ts.map +1 -0
- package/dist/media/frameCapture.js +145 -0
- package/dist/media/frameCapture.js.map +1 -0
- package/package.json +4 -4
|
@@ -4,13 +4,16 @@ var __decorate = (this && this.__decorate) || function (decorators, target, key,
|
|
|
4
4
|
else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
|
|
5
5
|
return c > 3 && r && Object.defineProperty(target, key, r), r;
|
|
6
6
|
};
|
|
7
|
+
var GeminiRealtimeClient_1;
|
|
7
8
|
import { RegisterClass } from '@memberjunction/global';
|
|
8
|
-
import {
|
|
9
|
+
import { RealtimeDiagLog, RealtimeToolBatchBarrier, ExtractToolSchedulingHint, } from '@memberjunction/ai';
|
|
10
|
+
import { GoogleGenAI, FunctionResponseScheduling, } from '@google/genai';
|
|
9
11
|
import { BaseRealtimeClient } from '../generic/baseRealtimeClient.js';
|
|
10
12
|
import { base64ToArrayBuffer } from '../audio/pcmUtils.js';
|
|
11
13
|
import { RealtimePcmPlayback } from '../audio/pcmPlayback.js';
|
|
12
14
|
import { RealtimeAudioMeter } from '../audio/audioMeter.js';
|
|
13
15
|
import { createPcmMicCapture } from '../audio/micCapture.js';
|
|
16
|
+
import { createStreamFrameCapture } from '../media/frameCapture.js';
|
|
14
17
|
// ── Audio constants (Gemini Live wire formats) ─────────────────────────────────
|
|
15
18
|
/** Gemini Live expects client audio as 16-bit signed PCM, 16 kHz, mono. */
|
|
16
19
|
const GEMINI_INPUT_SAMPLE_RATE = 16000;
|
|
@@ -18,6 +21,18 @@ const GEMINI_INPUT_SAMPLE_RATE = 16000;
|
|
|
18
21
|
const GEMINI_INPUT_AUDIO_MIME_TYPE = 'audio/pcm;rate=16000';
|
|
19
22
|
/** Gemini Live emits model audio as 16-bit signed PCM, 24 kHz, mono. */
|
|
20
23
|
const GEMINI_OUTPUT_SAMPLE_RATE = 24000;
|
|
24
|
+
// ── Legacy video-capability fallback ───────────────────────────────────────────
|
|
25
|
+
//
|
|
26
|
+
// Video capability and its rate ceiling are per-model data, minted from the provider's profile
|
|
27
|
+
// table into the session config. A mint from a server that predates those fields sends neither,
|
|
28
|
+
// and this client must still negotiate video for the models that had it — so these two values
|
|
29
|
+
// reproduce the behaviour that shipped before the fields existed, and NOTHING ELSE should read
|
|
30
|
+
// them. They are reachable only against an older server; delete both once no supported server
|
|
31
|
+
// mints a session config without `supportsInboundVideo`.
|
|
32
|
+
/** Model-id prefix that identified a video-capable Live model before the profile carried the flag. */
|
|
33
|
+
const LEGACY_VIDEO_MODEL_PREFIX = 'gemini-3.8-live';
|
|
34
|
+
/** The frame-rate ceiling this client hardcoded before `MaxInboundVideoRate` was minted. */
|
|
35
|
+
const LEGACY_VIDEO_MODEL_RATE = 1;
|
|
21
36
|
// ── Production playback engine ─────────────────────────────────────────────────
|
|
22
37
|
/**
|
|
23
38
|
* Web Audio playout scheduler for Gemini's 24 kHz PCM16 model audio.
|
|
@@ -88,13 +103,29 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
88
103
|
// ── Transport / audio resources ────────────────────────────────────────────
|
|
89
104
|
this.session = null;
|
|
90
105
|
this.micStream = null;
|
|
106
|
+
this.cameraStream = null;
|
|
91
107
|
this.micCapture = null;
|
|
108
|
+
this.cameraCapture = null;
|
|
92
109
|
this.playback = null;
|
|
110
|
+
this.firstVideoSendTimestamp = 0;
|
|
111
|
+
this.lastVideoSendTimestamp = 0;
|
|
112
|
+
this.videoFramesSent = 0;
|
|
113
|
+
this.resumptionHandle = null;
|
|
114
|
+
this.lastConnectArgs = null;
|
|
115
|
+
// ── Model capability & profile state ───────────────────────────────────────
|
|
116
|
+
this.idleSignal = 'turnComplete';
|
|
117
|
+
this.supportsScheduling = true;
|
|
118
|
+
this.supportsBlocking = true;
|
|
119
|
+
this.interactionInProgress = false;
|
|
120
|
+
this.toolBatchBarrier = new RealtimeToolBatchBarrier();
|
|
121
|
+
this.assistantSafetyBackstopTimer = null;
|
|
93
122
|
// ── Response state machine ─────────────────────────────────────────────────
|
|
94
123
|
/** Accumulates the in-flight assistant transcript across delta frames. */
|
|
95
124
|
this.pendingAssistantText = '';
|
|
96
125
|
/** Accumulates the in-flight user transcription across delta frames. */
|
|
97
126
|
this.pendingUserText = '';
|
|
127
|
+
/** Accumulates in-flight thought text deltas until finalized on turn completion. */
|
|
128
|
+
this.pendingThoughtText = '';
|
|
98
129
|
/** True while a model turn is in flight; gates (queues) client-triggered sends. */
|
|
99
130
|
this.responseActive = false;
|
|
100
131
|
/** The kind of the turn currently in flight; stamped at send time, reset on turnComplete. */
|
|
@@ -124,6 +155,34 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
124
155
|
*/
|
|
125
156
|
this.currentState = 'closed';
|
|
126
157
|
}
|
|
158
|
+
static { GeminiRealtimeClient_1 = this; }
|
|
159
|
+
static { this.ASSISTANT_SAFETY_BACKSTOP_MS = 15000; }
|
|
160
|
+
/** Returns the latest session resumption handle reported by the server, if any. */
|
|
161
|
+
get ResumptionHandle() {
|
|
162
|
+
return this.resumptionHandle;
|
|
163
|
+
}
|
|
164
|
+
/** Returns the count of video frames successfully sent over the established video track. */
|
|
165
|
+
get VideoFramesSent() {
|
|
166
|
+
return this.videoFramesSent;
|
|
167
|
+
}
|
|
168
|
+
/**
|
|
169
|
+
* Returns the cumulative active video duration in seconds across sent video frames.
|
|
170
|
+
* Represents the wall-clock span between first and last sent frames (span-not-sum) for stream telemetry.
|
|
171
|
+
* Provider-reported ImageTokens remains the authoritative financial billing basis.
|
|
172
|
+
*/
|
|
173
|
+
get VideoSeconds() {
|
|
174
|
+
if (this.firstVideoSendTimestamp === 0 || this.lastVideoSendTimestamp === 0) {
|
|
175
|
+
return 0;
|
|
176
|
+
}
|
|
177
|
+
return Math.max(1, Math.round((this.lastVideoSendTimestamp - this.firstVideoSendTimestamp) / 1000) + 1);
|
|
178
|
+
}
|
|
179
|
+
/**
|
|
180
|
+
* Whether the active model session enforces asynchronous non-blocking tool execution.
|
|
181
|
+
* Derived from model tooling capability (!supportsBlocking), separated from the idle signal (Reviewer Item 19).
|
|
182
|
+
*/
|
|
183
|
+
get isNonBlocking() {
|
|
184
|
+
return !this.supportsBlocking;
|
|
185
|
+
}
|
|
127
186
|
// ── BaseRealtimeClient: connection lifecycle ───────────────────────────────
|
|
128
187
|
/**
|
|
129
188
|
* Opens the client-direct Gemini Live session: creates the playout engine, connects with
|
|
@@ -131,21 +190,70 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
131
190
|
* values the server LOCKED into the token, so tampering is ignored by the API), then wires
|
|
132
191
|
* the mic-capture worklet. Reports `'listening'` once audio is flowing.
|
|
133
192
|
*/
|
|
134
|
-
|
|
193
|
+
/**
|
|
194
|
+
* Opens the client-direct Gemini Live session: creates the playout engine, connects with
|
|
195
|
+
* the ephemeral token + the server-built `SessionConfig` (`{ model, config }` — the same
|
|
196
|
+
* values the server LOCKED into the token, so tampering is ignored by the API), negotiates
|
|
197
|
+
* tracks, then wires the mic-capture worklet and optional video capture.
|
|
198
|
+
* Reports `'listening'` once audio is flowing.
|
|
199
|
+
*/
|
|
200
|
+
async Connect(config, micStream, cameraStream) {
|
|
135
201
|
this.micStream = micStream;
|
|
202
|
+
this.cameraStream = cameraStream ?? null;
|
|
203
|
+
this.clearSafetyBackstop();
|
|
204
|
+
this.toolBatchBarrier.Clear();
|
|
205
|
+
this.firstVideoSendTimestamp = 0;
|
|
206
|
+
this.lastVideoSendTimestamp = 0;
|
|
207
|
+
this.videoFramesSent = 0;
|
|
136
208
|
this.setState('connecting');
|
|
137
|
-
const { model, liveConfig } = this.parseSessionConfig(config);
|
|
209
|
+
const { model, liveConfig, idleSignal, supportsScheduling, supportsBlocking, supportsInboundVideo, maxInboundVideoRate, requestedTracks } = this.parseSessionConfig(config);
|
|
210
|
+
this.idleSignal = idleSignal;
|
|
211
|
+
this.supportsScheduling = supportsScheduling;
|
|
212
|
+
this.supportsBlocking = supportsBlocking;
|
|
213
|
+
// Negotiate tracks. Video capability and its frame-rate ceiling are PER-MODEL DATA, minted
|
|
214
|
+
// from the provider's profile table (GeminiLiveModelProfile.SupportsInboundVideo /
|
|
215
|
+
// .MaxInboundVideoRate) and carried in the session config. Deriving either from the model
|
|
216
|
+
// id would put a second answer to the same question in a second place: the two agreed only
|
|
217
|
+
// because the model names happened to line up, and the next model to break that pattern
|
|
218
|
+
// would diverge silently.
|
|
219
|
+
const isVideoModel = supportsInboundVideo ?? model.toLowerCase().startsWith(LEGACY_VIDEO_MODEL_PREFIX);
|
|
220
|
+
const supportedTracks = [
|
|
221
|
+
{ Modality: 'audio', Direction: 'inbound' },
|
|
222
|
+
{ Modality: 'audio', Direction: 'outbound' },
|
|
223
|
+
];
|
|
224
|
+
if (isVideoModel) {
|
|
225
|
+
supportedTracks.push({
|
|
226
|
+
Modality: 'video',
|
|
227
|
+
Direction: 'inbound',
|
|
228
|
+
Encoding: 'image/jpeg',
|
|
229
|
+
// The model's own ceiling. ResolveRequestedTracks takes the more restrictive of
|
|
230
|
+
// this and what the session requested, so this is what bounds the live track.
|
|
231
|
+
Rate: maxInboundVideoRate ?? LEGACY_VIDEO_MODEL_RATE,
|
|
232
|
+
UsageBasis: ['tokens', 'frames'],
|
|
233
|
+
RequiresConsent: true,
|
|
234
|
+
});
|
|
235
|
+
}
|
|
236
|
+
this.negotiateTracks(requestedTracks, supportedTracks);
|
|
138
237
|
this.playback = this.createPlayback();
|
|
139
|
-
|
|
238
|
+
const connectArgs = {
|
|
140
239
|
Model: model,
|
|
141
240
|
Config: liveConfig,
|
|
142
241
|
EphemeralToken: config.EphemeralToken,
|
|
143
242
|
OnMessage: (message) => this.handleServerMessage(message),
|
|
144
243
|
OnError: (event) => this.handleTransportError(event),
|
|
145
|
-
OnClose: () => this.handleTransportClose(),
|
|
146
|
-
}
|
|
244
|
+
OnClose: (event) => this.handleTransportClose(event),
|
|
245
|
+
};
|
|
246
|
+
this.lastConnectArgs = connectArgs;
|
|
247
|
+
this.session = await this.connectLiveSession(connectArgs);
|
|
147
248
|
this.setState('connected');
|
|
148
249
|
this.micCapture = await this.createMicCapture(micStream, (base64Pcm16) => this.sendMicChunk(base64Pcm16));
|
|
250
|
+
// Start camera capture if inbound video is established and cameraStream provided
|
|
251
|
+
if (this.cameraStream && this.IsTrackEstablished('video', 'inbound')) {
|
|
252
|
+
this.cameraCapture = createStreamFrameCapture(this.cameraStream, {
|
|
253
|
+
Rate: 1,
|
|
254
|
+
OnFrame: (frame) => this.SendVideoFrame(frame.data, frame.mimeType),
|
|
255
|
+
});
|
|
256
|
+
}
|
|
149
257
|
// Audio-activity capability (base obligation #9): agent side taps the playout
|
|
150
258
|
// engine's master gain; user side meters the mic stream. Null-safe — test fakes /
|
|
151
259
|
// no-WebAudio environments simply leave the session un-metered.
|
|
@@ -154,18 +262,28 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
154
262
|
this.setState('listening');
|
|
155
263
|
}
|
|
156
264
|
/**
|
|
157
|
-
* Tears down the session, mic capture, mic tracks, and playout engine,
|
|
158
|
-
* state machine, and emits a final `'closed'` (unless already `'error'`).
|
|
159
|
-
* more than once.
|
|
265
|
+
* Tears down the session, mic capture, mic tracks, camera capture, and playout engine,
|
|
266
|
+
* resets the response state machine, and emits a final `'closed'` (unless already `'error'`).
|
|
267
|
+
* Safe to call more than once.
|
|
160
268
|
*/
|
|
161
269
|
async Disconnect() {
|
|
162
270
|
this.closeAudioMeters();
|
|
271
|
+
this.clearSafetyBackstop();
|
|
272
|
+
this.toolBatchBarrier.Clear();
|
|
163
273
|
this.micStream?.getTracks().forEach((track) => track.stop());
|
|
164
274
|
this.micStream = null;
|
|
275
|
+
this.cameraCapture?.Stop();
|
|
276
|
+
this.cameraCapture = null;
|
|
277
|
+
this.cameraStream?.getTracks().forEach((track) => track.stop());
|
|
278
|
+
this.cameraStream = null;
|
|
165
279
|
this.micCapture?.Stop();
|
|
166
280
|
this.micCapture = null;
|
|
167
281
|
this.playback?.Close();
|
|
168
282
|
this.playback = null;
|
|
283
|
+
this.resumptionHandle = null;
|
|
284
|
+
this.firstVideoSendTimestamp = 0;
|
|
285
|
+
this.lastVideoSendTimestamp = 0;
|
|
286
|
+
this.videoFramesSent = 0;
|
|
169
287
|
if (this.session) {
|
|
170
288
|
try {
|
|
171
289
|
this.session.close();
|
|
@@ -202,6 +320,40 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
202
320
|
this.CancelActiveResponse();
|
|
203
321
|
this.enqueueOrRun(() => this.sendTriggeringUserTurn(text, 'normal', true));
|
|
204
322
|
}
|
|
323
|
+
/**
|
|
324
|
+
* Streams one base64 image frame over the established inbound video track.
|
|
325
|
+
*
|
|
326
|
+
* Enforces a 750ms minimum inter-frame spacing to serve as a backstop with deliberate jitter
|
|
327
|
+
* headroom for 1 fps (1000ms) pacers (such as `ChannelInboundVideoBridge`'s `setInterval`,
|
|
328
|
+
* `frameCapture`, and channel-level gates like `OnScreencastFrame`'s 1000ms pacer). Upstream
|
|
329
|
+
* cadence generators and channel gates are the primary enforcers of the nominal 1 fps ceiling,
|
|
330
|
+
* while this 750ms gate absorbs event loop and async dispatch jitter without dropping intended
|
|
331
|
+
* 1Hz frames, while preventing any unpaced callers from bursting above 1.33 fps.
|
|
332
|
+
*
|
|
333
|
+
* If inbound video is not established, returns `false` without error or frame sends (fallback).
|
|
334
|
+
*
|
|
335
|
+
* @returns `true` if the frame was dispatched to the session; `false` if dropped (throttled
|
|
336
|
+
* or track unestablished).
|
|
337
|
+
*/
|
|
338
|
+
SendVideoFrame(base64Image, mimeType = 'image/jpeg') {
|
|
339
|
+
if (!this.IsTrackEstablished('video', 'inbound')) {
|
|
340
|
+
return false;
|
|
341
|
+
}
|
|
342
|
+
const now = Date.now();
|
|
343
|
+
if (this.lastVideoSendTimestamp > 0 && now - this.lastVideoSendTimestamp < 750) {
|
|
344
|
+
return false; // Throttled: 750ms jitter headroom backstop for upstream 1 fps pacers (Reviewer Items 25, 30, 33)
|
|
345
|
+
}
|
|
346
|
+
this.lastVideoSendTimestamp = now;
|
|
347
|
+
if (this.firstVideoSendTimestamp === 0) {
|
|
348
|
+
this.firstVideoSendTimestamp = now;
|
|
349
|
+
}
|
|
350
|
+
this.videoFramesSent++;
|
|
351
|
+
// Reviewer Item 31: Send `video` alone — do not populate sibling `media` slot to avoid duplicate bytes & billing
|
|
352
|
+
this.session?.sendRealtimeInput({
|
|
353
|
+
video: { data: base64Image, mimeType },
|
|
354
|
+
});
|
|
355
|
+
return true;
|
|
356
|
+
}
|
|
205
357
|
/**
|
|
206
358
|
* @inheritdoc
|
|
207
359
|
*
|
|
@@ -218,11 +370,13 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
218
370
|
if (!this.session) {
|
|
219
371
|
return;
|
|
220
372
|
}
|
|
221
|
-
if (!this.responseActive && !this.IsAudioPlaying) {
|
|
373
|
+
if (!this.responseActive && !this.IsAudioPlaying && !this.interactionInProgress) {
|
|
222
374
|
return; // nothing active — no-op by contract
|
|
223
375
|
}
|
|
376
|
+
this.clearSafetyBackstop();
|
|
224
377
|
this.playback?.Flush();
|
|
225
378
|
this.responseActive = false;
|
|
379
|
+
this.interactionInProgress = false;
|
|
226
380
|
this.activeResponseKind = 'normal';
|
|
227
381
|
this.flushQueuedSends();
|
|
228
382
|
if (this.currentState === 'speaking') {
|
|
@@ -281,7 +435,12 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
281
435
|
return;
|
|
282
436
|
}
|
|
283
437
|
const name = this.pendingToolCallNames.get(callID) ?? '';
|
|
284
|
-
this.
|
|
438
|
+
if (this.isNonBlocking) {
|
|
439
|
+
this.sendToolResponseTurn(callID, name, outputJson);
|
|
440
|
+
}
|
|
441
|
+
else {
|
|
442
|
+
this.enqueueOrRun(() => this.sendToolResponseTurn(callID, name, outputJson));
|
|
443
|
+
}
|
|
285
444
|
}
|
|
286
445
|
/**
|
|
287
446
|
* Mutes / unmutes by toggling the mic tracks' `enabled` flag: the capture pipeline stays
|
|
@@ -297,7 +456,12 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
297
456
|
}
|
|
298
457
|
/** @inheritdoc */
|
|
299
458
|
get IsBusy() {
|
|
300
|
-
|
|
459
|
+
if (this.isNonBlocking) {
|
|
460
|
+
return (this.responseActive ||
|
|
461
|
+
this.interactionInProgress ||
|
|
462
|
+
!this.toolBatchBarrier.IsEmpty);
|
|
463
|
+
}
|
|
464
|
+
return this.responseActive || this.interactionInProgress;
|
|
301
465
|
}
|
|
302
466
|
/**
|
|
303
467
|
* @inheritdoc
|
|
@@ -354,21 +518,74 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
354
518
|
const model = typeof sessionConfig['model'] === 'string' ? sessionConfig['model'] : config.Model;
|
|
355
519
|
const raw = sessionConfig['config'];
|
|
356
520
|
const liveConfig = raw !== null && typeof raw === 'object' && !Array.isArray(raw) ? raw : {};
|
|
357
|
-
|
|
521
|
+
const rawIdle = sessionConfig['idleSignal'];
|
|
522
|
+
const idleSignal = rawIdle === 'interactionStatus' ? 'interactionStatus' : 'turnComplete';
|
|
523
|
+
const supportsScheduling = sessionConfig['supportsScheduling'] !== false;
|
|
524
|
+
const supportsBlocking = sessionConfig['supportsBlocking'] !== false;
|
|
525
|
+
// Per-model video legality, minted from the provider's profile table. Left undefined by a
|
|
526
|
+
// mint that predates these fields — see LEGACY_VIDEO_MODEL_PREFIX at the call site.
|
|
527
|
+
const supportsInboundVideo = typeof sessionConfig['supportsInboundVideo'] === 'boolean' ? sessionConfig['supportsInboundVideo'] : undefined;
|
|
528
|
+
const rawMaxVideoRate = sessionConfig['maxInboundVideoRate'];
|
|
529
|
+
const maxInboundVideoRate = typeof rawMaxVideoRate === 'number' && rawMaxVideoRate > 0 ? rawMaxVideoRate : undefined;
|
|
530
|
+
const rawRequestedTracks = sessionConfig['requestedTracks'];
|
|
531
|
+
let requestedTracks = undefined;
|
|
532
|
+
if (Array.isArray(rawRequestedTracks)) {
|
|
533
|
+
const list = [];
|
|
534
|
+
for (const item of rawRequestedTracks) {
|
|
535
|
+
if (item !== null && typeof item === 'object' && !Array.isArray(item)) {
|
|
536
|
+
const modality = typeof item['Modality'] === 'string' ? item['Modality'] : undefined;
|
|
537
|
+
const direction = item['Direction'];
|
|
538
|
+
if (modality && (direction === 'inbound' || direction === 'outbound')) {
|
|
539
|
+
list.push({
|
|
540
|
+
Modality: modality,
|
|
541
|
+
Direction: direction,
|
|
542
|
+
Encoding: typeof item['Encoding'] === 'string' ? item['Encoding'] : undefined,
|
|
543
|
+
Rate: typeof item['Rate'] === 'number' ? item['Rate'] : undefined,
|
|
544
|
+
RequiresConsent: typeof item['RequiresConsent'] === 'boolean' ? item['RequiresConsent'] : undefined,
|
|
545
|
+
});
|
|
546
|
+
}
|
|
547
|
+
}
|
|
548
|
+
}
|
|
549
|
+
requestedTracks = list;
|
|
550
|
+
}
|
|
551
|
+
return { model, liveConfig, idleSignal, supportsScheduling, supportsBlocking, supportsInboundVideo, maxInboundVideoRate, requestedTracks };
|
|
358
552
|
}
|
|
359
|
-
/** Streams one base64 PCM16 mic chunk to the model (no-op once the session is gone). */
|
|
553
|
+
/** Streams one base64 PCM16 mic chunk to the model (no-op once the session is gone, closed, or in error). */
|
|
360
554
|
sendMicChunk(base64Pcm16) {
|
|
361
|
-
this.session
|
|
362
|
-
|
|
363
|
-
}
|
|
555
|
+
if (!this.session || this.currentState === 'closed' || this.currentState === 'error') {
|
|
556
|
+
return;
|
|
557
|
+
}
|
|
558
|
+
try {
|
|
559
|
+
this.session.sendRealtimeInput({
|
|
560
|
+
audio: { data: base64Pcm16, mimeType: GEMINI_INPUT_AUDIO_MIME_TYPE },
|
|
561
|
+
});
|
|
562
|
+
}
|
|
563
|
+
catch (err) {
|
|
564
|
+
RealtimeDiagLog(`[GeminiRealtimeClient] sendMicChunk failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
565
|
+
}
|
|
364
566
|
}
|
|
365
567
|
/** Surfaces a fatal websocket error and marks the session unusable. */
|
|
366
568
|
handleTransportError(event) {
|
|
367
|
-
|
|
569
|
+
const detail = event.message || (event.error instanceof Error ? event.error.message : String(event.error ?? 'unknown'));
|
|
570
|
+
RealtimeDiagLog(`[GeminiRealtimeClient] Transport error: ${detail}`);
|
|
571
|
+
this.emitError({ Message: `Gemini Live transport error: ${detail}`, Fatal: true });
|
|
368
572
|
this.setState('error');
|
|
369
573
|
}
|
|
370
574
|
/** Reflects a provider-side close (unless the session already ended in error). */
|
|
371
|
-
handleTransportClose() {
|
|
575
|
+
handleTransportClose(event) {
|
|
576
|
+
const code = event?.code;
|
|
577
|
+
const reason = event?.reason;
|
|
578
|
+
const wasClean = event?.wasClean;
|
|
579
|
+
RealtimeDiagLog(`[GeminiRealtimeClient] Transport closed: code=${code} reason=${reason} wasClean=${wasClean}`);
|
|
580
|
+
const isAbnormal = (code !== undefined && code !== 0 && code !== 1000 && code !== 1005) || (wasClean === false && code !== 1000 && code !== 0 && code !== 1005 && code !== undefined);
|
|
581
|
+
if (isAbnormal && this.currentState !== 'error') {
|
|
582
|
+
this.emitError({
|
|
583
|
+
Message: `Gemini Live connection closed (${code}): ${reason || 'unexpected disconnect'}`,
|
|
584
|
+
Fatal: true,
|
|
585
|
+
});
|
|
586
|
+
this.setState('error');
|
|
587
|
+
return;
|
|
588
|
+
}
|
|
372
589
|
if (this.currentState !== 'error' && this.currentState !== 'closed') {
|
|
373
590
|
this.setState('closed');
|
|
374
591
|
}
|
|
@@ -379,6 +596,22 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
379
596
|
* handlers, including `usageMetadata` → {@link emitUsage}.
|
|
380
597
|
*/
|
|
381
598
|
handleServerMessage(message) {
|
|
599
|
+
this.checkInteractionStatus(message);
|
|
600
|
+
// Session continuity: track resumption token updates (F7)
|
|
601
|
+
if (message.sessionResumptionUpdate) {
|
|
602
|
+
if (message.sessionResumptionUpdate.resumable === false) {
|
|
603
|
+
this.resumptionHandle = null;
|
|
604
|
+
}
|
|
605
|
+
else if (message.sessionResumptionUpdate.newHandle) {
|
|
606
|
+
this.resumptionHandle = message.sessionResumptionUpdate.newHandle;
|
|
607
|
+
}
|
|
608
|
+
}
|
|
609
|
+
// Server approaching timeout / abort: reconnect seamlessly using resumption handle (F7)
|
|
610
|
+
if (message.goAway) {
|
|
611
|
+
if (this.resumptionHandle) {
|
|
612
|
+
void this.resumeSession(this.resumptionHandle);
|
|
613
|
+
}
|
|
614
|
+
}
|
|
382
615
|
if (message.serverContent) {
|
|
383
616
|
this.handleServerContent(message.serverContent);
|
|
384
617
|
}
|
|
@@ -389,6 +622,68 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
389
622
|
this.handleUsageMetadata(message.usageMetadata);
|
|
390
623
|
}
|
|
391
624
|
}
|
|
625
|
+
/**
|
|
626
|
+
* Resumes the live session using a previously captured session resumption handle (F7).
|
|
627
|
+
*/
|
|
628
|
+
async resumeSession(handle) {
|
|
629
|
+
if (!this.lastConnectArgs) {
|
|
630
|
+
return;
|
|
631
|
+
}
|
|
632
|
+
try {
|
|
633
|
+
const reconnectArgs = {
|
|
634
|
+
...this.lastConnectArgs,
|
|
635
|
+
Config: {
|
|
636
|
+
...this.lastConnectArgs.Config,
|
|
637
|
+
sessionResumption: { handle },
|
|
638
|
+
},
|
|
639
|
+
};
|
|
640
|
+
const oldSession = this.session;
|
|
641
|
+
const newSession = await this.connectLiveSession(reconnectArgs);
|
|
642
|
+
this.session = newSession;
|
|
643
|
+
this.lastConnectArgs = reconnectArgs;
|
|
644
|
+
try {
|
|
645
|
+
oldSession?.close();
|
|
646
|
+
}
|
|
647
|
+
catch {
|
|
648
|
+
// The old socket has already been replaced by newSession, so a close failure on it cannot affect the new session
|
|
649
|
+
}
|
|
650
|
+
}
|
|
651
|
+
catch (err) {
|
|
652
|
+
RealtimeDiagLog(`[GeminiRealtimeClient] Session resumption failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
653
|
+
}
|
|
654
|
+
}
|
|
655
|
+
/**
|
|
656
|
+
* Inspects inbound frames for the untyped `interaction_status` / `interactionStatus` wire field
|
|
657
|
+
* documented for Gemini Live Extended Thinking (V3). Narrowed via null-safe object/string helpers.
|
|
658
|
+
*/
|
|
659
|
+
checkInteractionStatus(message) {
|
|
660
|
+
const msgObj = GeminiRealtimeClient_1.readObject(message);
|
|
661
|
+
const contentObj = GeminiRealtimeClient_1.readObject(message.serverContent);
|
|
662
|
+
const rawStatus = GeminiRealtimeClient_1.readString(msgObj?.['interaction_status']) ??
|
|
663
|
+
GeminiRealtimeClient_1.readString(msgObj?.['interactionStatus']) ??
|
|
664
|
+
GeminiRealtimeClient_1.readString(contentObj?.['interaction_status']) ??
|
|
665
|
+
GeminiRealtimeClient_1.readString(contentObj?.['interactionStatus']);
|
|
666
|
+
if (!rawStatus) {
|
|
667
|
+
return;
|
|
668
|
+
}
|
|
669
|
+
const status = rawStatus.toUpperCase();
|
|
670
|
+
if (status === 'IN_PROGRESS') {
|
|
671
|
+
this.interactionInProgress = true;
|
|
672
|
+
this.responseActive = true;
|
|
673
|
+
if (this.idleSignal === 'interactionStatus') {
|
|
674
|
+
this.scheduleSafetyBackstop();
|
|
675
|
+
}
|
|
676
|
+
}
|
|
677
|
+
else if (status === 'IDLE') {
|
|
678
|
+
this.clearSafetyBackstop();
|
|
679
|
+
if (this.idleSignal === 'interactionStatus') {
|
|
680
|
+
this.handleIdleTerminal();
|
|
681
|
+
}
|
|
682
|
+
else {
|
|
683
|
+
this.interactionInProgress = false;
|
|
684
|
+
}
|
|
685
|
+
}
|
|
686
|
+
}
|
|
392
687
|
/**
|
|
393
688
|
* Emits a usage update from a server message's `usageMetadata`.
|
|
394
689
|
*
|
|
@@ -400,9 +695,53 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
400
695
|
* the server-bridged `GeminiRealtime` driver forwards the same payload to `IRealtimeSession.OnUsage`.
|
|
401
696
|
*/
|
|
402
697
|
handleUsageMetadata(usageMetadata) {
|
|
698
|
+
let inputDetails;
|
|
699
|
+
if (usageMetadata.promptTokensDetails && Array.isArray(usageMetadata.promptTokensDetails)) {
|
|
700
|
+
for (const detail of usageMetadata.promptTokensDetails) {
|
|
701
|
+
if (typeof detail.tokenCount === 'number') {
|
|
702
|
+
inputDetails = inputDetails ?? {};
|
|
703
|
+
const mod = String(detail.modality ?? '').toUpperCase();
|
|
704
|
+
if (mod === 'AUDIO') {
|
|
705
|
+
inputDetails.AudioTokens = (inputDetails.AudioTokens ?? 0) + detail.tokenCount;
|
|
706
|
+
}
|
|
707
|
+
else if (mod === 'TEXT') {
|
|
708
|
+
inputDetails.TextTokens = (inputDetails.TextTokens ?? 0) + detail.tokenCount;
|
|
709
|
+
}
|
|
710
|
+
else if (mod === 'IMAGE') {
|
|
711
|
+
inputDetails.ImageTokens = (inputDetails.ImageTokens ?? 0) + detail.tokenCount;
|
|
712
|
+
}
|
|
713
|
+
}
|
|
714
|
+
}
|
|
715
|
+
}
|
|
716
|
+
/**
|
|
717
|
+
* Cost Attribution Note (F6 & Reviewer Item 29):
|
|
718
|
+
* Inbound video frames are sent as individual JPEG images (V5) and billed on the video pricing tier
|
|
719
|
+
* ($0.002 / min, or $1.00 / 1M tokens). The Gemini Live API reports token consumption via
|
|
720
|
+
* usageMetadata.promptTokensDetails partitioned into AUDIO, TEXT, and IMAGE (where video frame tokens
|
|
721
|
+
* are accounted under IMAGE).
|
|
722
|
+
*
|
|
723
|
+
* We expose two candidate cost and telemetry signals:
|
|
724
|
+
* 1. Authoritative Vendor Signal: InputTokenDetails.ImageTokens from promptTokensDetails represents
|
|
725
|
+
* the actual token consumption billed by the Google inference provider.
|
|
726
|
+
* 2. Track-Level Video Telemetry: VideoFrames (cumulative frames sent) and VideoSeconds (cumulative
|
|
727
|
+
* active video duration) client-side counters provide fine-grained telemetry and rate attribution.
|
|
728
|
+
*
|
|
729
|
+
* Logging both signals enables operational drift detection: divergence between client-sent VideoFrames
|
|
730
|
+
* and provider-received ImageTokens immediately surfaces frame drops or network throttling in production.
|
|
731
|
+
*/
|
|
732
|
+
if (this.videoFramesSent > 0) {
|
|
733
|
+
inputDetails = inputDetails ?? {};
|
|
734
|
+
inputDetails.VideoFrames = this.videoFramesSent;
|
|
735
|
+
inputDetails.VideoSeconds = this.VideoSeconds;
|
|
736
|
+
}
|
|
403
737
|
this.emitUsage({
|
|
404
738
|
InputTokens: typeof usageMetadata.promptTokenCount === 'number' ? usageMetadata.promptTokenCount : undefined,
|
|
405
739
|
OutputTokens: typeof usageMetadata.responseTokenCount === 'number' ? usageMetadata.responseTokenCount : undefined,
|
|
740
|
+
...(inputDetails ? { InputTokenDetails: inputDetails } : {}),
|
|
741
|
+
...(this.videoFramesSent > 0 ? {
|
|
742
|
+
VideoFrames: this.videoFramesSent,
|
|
743
|
+
VideoSeconds: this.VideoSeconds,
|
|
744
|
+
} : {}),
|
|
406
745
|
Raw: usageMetadata,
|
|
407
746
|
});
|
|
408
747
|
}
|
|
@@ -413,6 +752,7 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
413
752
|
}
|
|
414
753
|
if (content.modelTurn) {
|
|
415
754
|
this.handleModelAudio(content.modelTurn);
|
|
755
|
+
this.handleModelThoughts(content.modelTurn);
|
|
416
756
|
}
|
|
417
757
|
if (content.inputTranscription) {
|
|
418
758
|
this.handleUserTranscription(content.inputTranscription);
|
|
@@ -420,10 +760,25 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
420
760
|
if (content.outputTranscription) {
|
|
421
761
|
this.handleAssistantTranscription(content.outputTranscription);
|
|
422
762
|
}
|
|
763
|
+
if (content.generationComplete) {
|
|
764
|
+
this.handleGenerationComplete();
|
|
765
|
+
}
|
|
423
766
|
if (content.turnComplete) {
|
|
424
767
|
this.handleTurnComplete();
|
|
425
768
|
}
|
|
426
769
|
}
|
|
770
|
+
/**
|
|
771
|
+
* generationComplete: indicates the model has finished generating all tokens for the turn.
|
|
772
|
+
* Playout may still be active (the delay between generationComplete and turnComplete).
|
|
773
|
+
*
|
|
774
|
+
* Per Reviewer Item 20: Draining the queue happens on turnComplete or true IDLE, not prematurely
|
|
775
|
+
* on generationComplete. We set responseActive = false so busy state reflects token completion.
|
|
776
|
+
*/
|
|
777
|
+
handleGenerationComplete() {
|
|
778
|
+
if (this.idleSignal === 'turnComplete') {
|
|
779
|
+
this.responseActive = false;
|
|
780
|
+
}
|
|
781
|
+
}
|
|
427
782
|
/**
|
|
428
783
|
* Barge-in: the provider stopped generating because the user spoke. Flush every scheduled
|
|
429
784
|
* playout source (per the Live API contract, `interrupted` is the signal to empty the
|
|
@@ -434,6 +789,7 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
434
789
|
*/
|
|
435
790
|
handleInterruption() {
|
|
436
791
|
this.playback?.Flush();
|
|
792
|
+
this.finalizeThoughtTranscript();
|
|
437
793
|
this.emitInterruption();
|
|
438
794
|
this.setState('listening');
|
|
439
795
|
}
|
|
@@ -443,6 +799,9 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
443
799
|
return;
|
|
444
800
|
}
|
|
445
801
|
for (const part of modelTurn.parts) {
|
|
802
|
+
if (part.thought) {
|
|
803
|
+
continue; // Thoughts are reasoning summaries, never spoken audio
|
|
804
|
+
}
|
|
446
805
|
const data = part.inlineData?.data;
|
|
447
806
|
if (data) {
|
|
448
807
|
this.markGenerationStarted();
|
|
@@ -450,6 +809,28 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
450
809
|
}
|
|
451
810
|
}
|
|
452
811
|
}
|
|
812
|
+
/**
|
|
813
|
+
* Extracts thought parts (`part.thought === true`) from model turns and emits them
|
|
814
|
+
* as narration transcript deltas (`Kind: 'narration'`). Thought summaries are reasoning
|
|
815
|
+
* notes, never synthesized as assistant speech.
|
|
816
|
+
*/
|
|
817
|
+
handleModelThoughts(modelTurn) {
|
|
818
|
+
if (!modelTurn.parts) {
|
|
819
|
+
return;
|
|
820
|
+
}
|
|
821
|
+
for (const part of modelTurn.parts) {
|
|
822
|
+
if (part.thought && part.text) {
|
|
823
|
+
this.pendingThoughtText += part.text;
|
|
824
|
+
this.emitTranscript({
|
|
825
|
+
Role: 'Assistant',
|
|
826
|
+
Text: part.text,
|
|
827
|
+
IsFinal: false,
|
|
828
|
+
Kind: 'narration',
|
|
829
|
+
IsThought: true,
|
|
830
|
+
});
|
|
831
|
+
}
|
|
832
|
+
}
|
|
833
|
+
}
|
|
453
834
|
/**
|
|
454
835
|
* User transcription: each frame's `text` is an incremental DELTA (emitted with
|
|
455
836
|
* `IsFinal: false`); the accumulated turn text is finalized on the `finished` flag — or,
|
|
@@ -487,39 +868,78 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
487
868
|
}
|
|
488
869
|
/**
|
|
489
870
|
* Surfaces the model's tool calls to the host and caches each callID→name for
|
|
490
|
-
* {@link SendToolResult}.
|
|
491
|
-
*
|
|
492
|
-
*
|
|
493
|
-
*
|
|
494
|
-
*
|
|
871
|
+
* {@link SendToolResult}.
|
|
872
|
+
*
|
|
873
|
+
* In synchronous BLOCKING mode (e.g. 3.1 preview), the model yields the floor pending
|
|
874
|
+
* the tool result: `responseActive` is cleared so the tool response can take the floor,
|
|
875
|
+
* and the client silently leaves 'speaking'.
|
|
876
|
+
*
|
|
877
|
+
* In asynchronous NON_BLOCKING mode (Extended Thinking and 3.8 default), the model keeps
|
|
878
|
+
* generating and reasoning in the background. We must NOT set `responseActive = false`,
|
|
879
|
+
* as the model is not idle and still generating. The tool batch barrier tracks the call.
|
|
495
880
|
*/
|
|
496
881
|
handleToolCallFrame(functionCalls) {
|
|
497
882
|
if (!functionCalls || functionCalls.length === 0) {
|
|
498
883
|
return;
|
|
499
884
|
}
|
|
500
|
-
if (this.
|
|
501
|
-
this.currentState
|
|
885
|
+
if (!this.isNonBlocking) {
|
|
886
|
+
if (this.currentState === 'speaking') {
|
|
887
|
+
this.currentState = 'connected';
|
|
888
|
+
}
|
|
889
|
+
this.responseActive = false;
|
|
502
890
|
}
|
|
503
|
-
this.responseActive = false;
|
|
504
891
|
for (const call of functionCalls) {
|
|
505
892
|
const callID = call.id ?? '';
|
|
506
893
|
const toolName = call.name ?? '';
|
|
507
894
|
this.pendingToolCallNames.set(callID, toolName);
|
|
895
|
+
this.toolBatchBarrier.TrackPendingCall(callID, () => {
|
|
896
|
+
this.handleToolBatchTimeout();
|
|
897
|
+
});
|
|
508
898
|
this.emitToolCall({ CallID: callID, ToolName: toolName, ArgumentsJson: JSON.stringify(call.args ?? {}) });
|
|
509
899
|
}
|
|
510
900
|
}
|
|
511
901
|
/**
|
|
512
|
-
* Turn boundary: finalize any un-finished assistant transcript
|
|
513
|
-
*
|
|
514
|
-
*
|
|
902
|
+
* Turn boundary: finalize any un-finished assistant transcript.
|
|
903
|
+
* Under 'turnComplete' idle signal, release the busy lock, reset response kind,
|
|
904
|
+
* drain queued sends, and return the floor to the user.
|
|
905
|
+
* Under 'interactionStatus' (Extended Thinking), turnComplete does NOT indicate idle:
|
|
906
|
+
* background reasoning or async tool calls may still be in flight, so the busy lock
|
|
907
|
+
* and queued sends remain held until the true IDLE signal lands.
|
|
515
908
|
*/
|
|
516
909
|
handleTurnComplete() {
|
|
517
910
|
this.finalizeAssistantTranscript();
|
|
911
|
+
this.finalizeThoughtTranscript();
|
|
912
|
+
if (this.idleSignal === 'turnComplete') {
|
|
913
|
+
this.responseActive = false;
|
|
914
|
+
this.activeResponseKind = 'normal';
|
|
915
|
+
this.openClientTurn = false; // the completed generation consumed any open client content
|
|
916
|
+
this.flushQueuedSends();
|
|
917
|
+
if (this.currentState === 'speaking') {
|
|
918
|
+
this.setState('listening');
|
|
919
|
+
}
|
|
920
|
+
}
|
|
921
|
+
else {
|
|
922
|
+
// Extended Thinking: turn complete within active interaction — re-arm liveness guard
|
|
923
|
+
this.scheduleSafetyBackstop();
|
|
924
|
+
}
|
|
925
|
+
}
|
|
926
|
+
/**
|
|
927
|
+
* Reached when the server emits true IDLE (interactionStatus) or the safety backstop fires.
|
|
928
|
+
* Releases busy state, commits any deferred open client turns without cutting off generation,
|
|
929
|
+
* drains queued sends, and returns the floor.
|
|
930
|
+
*/
|
|
931
|
+
handleIdleTerminal() {
|
|
932
|
+
this.clearSafetyBackstop();
|
|
933
|
+
this.interactionInProgress = false;
|
|
518
934
|
this.responseActive = false;
|
|
519
935
|
this.activeResponseKind = 'normal';
|
|
520
|
-
this.
|
|
936
|
+
this.finalizeThoughtTranscript();
|
|
937
|
+
if (this.openClientTurn) {
|
|
938
|
+
this.session?.sendClientContent({ turnComplete: true });
|
|
939
|
+
this.openClientTurn = false;
|
|
940
|
+
}
|
|
521
941
|
this.flushQueuedSends();
|
|
522
|
-
if (this.currentState === 'speaking') {
|
|
942
|
+
if (this.currentState === 'speaking' && !this.IsAudioPlaying) {
|
|
523
943
|
this.setState('listening');
|
|
524
944
|
}
|
|
525
945
|
}
|
|
@@ -531,6 +951,10 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
531
951
|
markGenerationStarted() {
|
|
532
952
|
this.finalizeUserTranscript();
|
|
533
953
|
this.responseActive = true;
|
|
954
|
+
if (this.idleSignal === 'interactionStatus') {
|
|
955
|
+
this.interactionInProgress = true;
|
|
956
|
+
this.scheduleSafetyBackstop();
|
|
957
|
+
}
|
|
534
958
|
if (this.currentState !== 'speaking') {
|
|
535
959
|
this.setState('speaking');
|
|
536
960
|
}
|
|
@@ -551,6 +975,14 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
551
975
|
this.emitTranscript({ Role: 'Assistant', Text: text, IsFinal: true, Kind: this.activeResponseKind });
|
|
552
976
|
}
|
|
553
977
|
}
|
|
978
|
+
/** Emits the accumulated thought turn as final (if non-empty) with Kind: 'narration' and IsThought: true. */
|
|
979
|
+
finalizeThoughtTranscript() {
|
|
980
|
+
const text = this.pendingThoughtText;
|
|
981
|
+
this.pendingThoughtText = '';
|
|
982
|
+
if (text.trim().length > 0) {
|
|
983
|
+
this.emitTranscript({ Role: 'Assistant', Text: text, IsFinal: true, Kind: 'narration', IsThought: true });
|
|
984
|
+
}
|
|
985
|
+
}
|
|
554
986
|
// ── Collision-safe send machinery ──────────────────────────────────────────
|
|
555
987
|
/**
|
|
556
988
|
* Runs a send immediately when no turn is in flight; otherwise queues it for the next
|
|
@@ -595,6 +1027,10 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
595
1027
|
// `openClientTurn` is left as-is — the model `turnComplete` that follows clears it.
|
|
596
1028
|
session.sendRealtimeInput({ text });
|
|
597
1029
|
this.responseActive = true;
|
|
1030
|
+
if (this.idleSignal === 'interactionStatus') {
|
|
1031
|
+
this.interactionInProgress = true;
|
|
1032
|
+
this.scheduleSafetyBackstop();
|
|
1033
|
+
}
|
|
598
1034
|
this.activeResponseKind = kind;
|
|
599
1035
|
if (emitSpeaking) {
|
|
600
1036
|
this.setState('speaking');
|
|
@@ -603,28 +1039,68 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
603
1039
|
/**
|
|
604
1040
|
* Sends the tool response (Gemini continues the turn with it) and marks the model busy.
|
|
605
1041
|
*
|
|
606
|
-
*
|
|
607
|
-
* tool response alone does NOT start generation — the Live API
|
|
608
|
-
* until the turn is committed. The empty-turn commit
|
|
609
|
-
* `sendClientContent({ turnComplete: true })`
|
|
610
|
-
*
|
|
1042
|
+
* In BLOCKING mode (legacy 3.1), if context notes have left a client content turn open
|
|
1043
|
+
* ({@link openClientTurn}), the tool response alone does NOT start generation — the Live API
|
|
1044
|
+
* holds for more client input until the turn is committed. The empty-turn commit
|
|
1045
|
+
* (`sendClientContent({ turnComplete: true })`) releases generation.
|
|
1046
|
+
*
|
|
1047
|
+
* In NON_BLOCKING mode (Extended Thinking and 3.8 default), setting `turnComplete: true`
|
|
1048
|
+
* unconditionally interrupts active model generation mid-sentence. We must NOT commit
|
|
1049
|
+
* openClientTurn here; it is deferred until true IDLE (or a subsequent user turn commit).
|
|
611
1050
|
*/
|
|
1051
|
+
/**
|
|
1052
|
+
* Extracts scheduling hints (`__mj_scheduling` or legacy `scheduling`) from the tool output,
|
|
1053
|
+
* strips both keys so they do not leak into the model's response payload, resolves the scheduling
|
|
1054
|
+
* directive accepting both 'INTERRUPT' and 'INTERRUPTED', and warns on unrecognized values or
|
|
1055
|
+
* unsupported models (delegates normalization to Core `ExtractToolSchedulingHint`).
|
|
1056
|
+
*/
|
|
1057
|
+
resolveFunctionScheduling(parsed, toolName) {
|
|
1058
|
+
const hint = ExtractToolSchedulingHint(parsed, toolName, 'GeminiRealtimeClient');
|
|
1059
|
+
if (!hint) {
|
|
1060
|
+
return undefined;
|
|
1061
|
+
}
|
|
1062
|
+
const schedStr = hint === 'silent' ? 'SILENT' : hint === 'whenIdle' ? 'WHEN_IDLE' : 'INTERRUPT';
|
|
1063
|
+
if (!this.supportsScheduling) {
|
|
1064
|
+
console.warn(`[GeminiRealtimeClient] Dropping scheduling hint "${schedStr}" for tool "${toolName}": ` +
|
|
1065
|
+
`function scheduling is only supported on gemini-3.8-live.`);
|
|
1066
|
+
return undefined;
|
|
1067
|
+
}
|
|
1068
|
+
switch (hint) {
|
|
1069
|
+
case 'silent':
|
|
1070
|
+
return FunctionResponseScheduling.SILENT;
|
|
1071
|
+
case 'whenIdle':
|
|
1072
|
+
return FunctionResponseScheduling.WHEN_IDLE;
|
|
1073
|
+
case 'interrupt':
|
|
1074
|
+
return FunctionResponseScheduling.INTERRUPT;
|
|
1075
|
+
}
|
|
1076
|
+
}
|
|
612
1077
|
sendToolResponseTurn(callID, name, outputJson) {
|
|
613
1078
|
const session = this.session;
|
|
614
1079
|
if (!session) {
|
|
615
1080
|
return;
|
|
616
1081
|
}
|
|
1082
|
+
this.toolBatchBarrier.RecordResult(callID);
|
|
1083
|
+
const parsed = this.parseToolOutput(outputJson);
|
|
1084
|
+
const sched = this.resolveFunctionScheduling(parsed, name);
|
|
1085
|
+
const functionResponse = {
|
|
1086
|
+
id: callID,
|
|
1087
|
+
name,
|
|
1088
|
+
response: parsed,
|
|
1089
|
+
...(sched ? { scheduling: sched } : {}),
|
|
1090
|
+
};
|
|
617
1091
|
session.sendToolResponse({
|
|
618
|
-
functionResponses: [
|
|
1092
|
+
functionResponses: [functionResponse],
|
|
619
1093
|
});
|
|
620
|
-
if (this.
|
|
621
|
-
|
|
622
|
-
|
|
1094
|
+
if (!this.isNonBlocking) {
|
|
1095
|
+
if (this.openClientTurn) {
|
|
1096
|
+
session.sendClientContent({ turnComplete: true });
|
|
1097
|
+
this.openClientTurn = false;
|
|
1098
|
+
}
|
|
1099
|
+
this.responseActive = true;
|
|
1100
|
+
this.activeResponseKind = 'normal';
|
|
1101
|
+
this.setState('speaking');
|
|
623
1102
|
}
|
|
624
1103
|
this.pendingToolCallNames.delete(callID);
|
|
625
|
-
this.responseActive = true;
|
|
626
|
-
this.activeResponseKind = 'normal';
|
|
627
|
-
this.setState('speaking');
|
|
628
1104
|
}
|
|
629
1105
|
/**
|
|
630
1106
|
* Parses a JSON-stringified tool result into the structured object Gemini's
|
|
@@ -644,23 +1120,60 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
|
|
|
644
1120
|
}
|
|
645
1121
|
}
|
|
646
1122
|
// ── Helpers ────────────────────────────────────────────────────────────────
|
|
1123
|
+
scheduleSafetyBackstop(delayMs = GeminiRealtimeClient_1.ASSISTANT_SAFETY_BACKSTOP_MS) {
|
|
1124
|
+
this.clearSafetyBackstop();
|
|
1125
|
+
this.assistantSafetyBackstopTimer = setTimeout(() => {
|
|
1126
|
+
this.assistantSafetyBackstopTimer = null;
|
|
1127
|
+
this.handleSafetyBackstopTrigger();
|
|
1128
|
+
}, delayMs);
|
|
1129
|
+
}
|
|
1130
|
+
clearSafetyBackstop() {
|
|
1131
|
+
if (this.assistantSafetyBackstopTimer) {
|
|
1132
|
+
clearTimeout(this.assistantSafetyBackstopTimer);
|
|
1133
|
+
this.assistantSafetyBackstopTimer = null;
|
|
1134
|
+
}
|
|
1135
|
+
}
|
|
1136
|
+
handleSafetyBackstopTrigger() {
|
|
1137
|
+
this.handleIdleTerminal();
|
|
1138
|
+
}
|
|
1139
|
+
handleToolBatchTimeout() {
|
|
1140
|
+
if (this.idleSignal === 'turnComplete') {
|
|
1141
|
+
this.flushQueuedSends();
|
|
1142
|
+
}
|
|
1143
|
+
}
|
|
647
1144
|
/** Resets the per-session response state machine (used on Disconnect). */
|
|
648
1145
|
resetResponseState() {
|
|
649
1146
|
this.pendingAssistantText = '';
|
|
650
1147
|
this.pendingUserText = '';
|
|
1148
|
+
this.pendingThoughtText = '';
|
|
651
1149
|
this.responseActive = false;
|
|
1150
|
+
this.interactionInProgress = false;
|
|
652
1151
|
this.activeResponseKind = 'normal';
|
|
653
1152
|
this.openClientTurn = false;
|
|
654
1153
|
this.queuedSends = [];
|
|
655
1154
|
this.pendingToolCallNames.clear();
|
|
1155
|
+
this.toolBatchBarrier.Clear();
|
|
1156
|
+
this.clearSafetyBackstop();
|
|
656
1157
|
}
|
|
657
1158
|
/** Updates the client's own state view and emits the change to the host. */
|
|
658
1159
|
setState(state) {
|
|
659
1160
|
this.currentState = state;
|
|
660
1161
|
this.emitStateChange(state);
|
|
661
1162
|
}
|
|
1163
|
+
static readObject(value) {
|
|
1164
|
+
return value !== null && typeof value === 'object' && !Array.isArray(value)
|
|
1165
|
+
? value
|
|
1166
|
+
: undefined;
|
|
1167
|
+
}
|
|
1168
|
+
static readString(value) {
|
|
1169
|
+
if (typeof value !== 'string') {
|
|
1170
|
+
return undefined;
|
|
1171
|
+
}
|
|
1172
|
+
const trimmed = value.trim();
|
|
1173
|
+
return trimmed.length > 0 ? trimmed : undefined;
|
|
1174
|
+
}
|
|
662
1175
|
};
|
|
663
|
-
GeminiRealtimeClient = __decorate([
|
|
1176
|
+
GeminiRealtimeClient = GeminiRealtimeClient_1 = __decorate([
|
|
664
1177
|
RegisterClass(BaseRealtimeClient, 'gemini')
|
|
665
1178
|
], GeminiRealtimeClient);
|
|
666
1179
|
export { GeminiRealtimeClient };
|