@memberjunction/ai-realtime-client 6.1.1 → 6.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,13 +4,16 @@ var __decorate = (this && this.__decorate) || function (decorators, target, key,
4
4
  else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
5
5
  return c > 3 && r && Object.defineProperty(target, key, r), r;
6
6
  };
7
+ var GeminiRealtimeClient_1;
7
8
  import { RegisterClass } from '@memberjunction/global';
8
- import { GoogleGenAI, } from '@google/genai';
9
+ import { RealtimeDiagLog, RealtimeToolBatchBarrier, ExtractToolSchedulingHint, } from '@memberjunction/ai';
10
+ import { GoogleGenAI, FunctionResponseScheduling, } from '@google/genai';
9
11
  import { BaseRealtimeClient } from '../generic/baseRealtimeClient.js';
10
12
  import { base64ToArrayBuffer } from '../audio/pcmUtils.js';
11
13
  import { RealtimePcmPlayback } from '../audio/pcmPlayback.js';
12
14
  import { RealtimeAudioMeter } from '../audio/audioMeter.js';
13
15
  import { createPcmMicCapture } from '../audio/micCapture.js';
16
+ import { createStreamFrameCapture } from '../media/frameCapture.js';
14
17
  // ── Audio constants (Gemini Live wire formats) ─────────────────────────────────
15
18
  /** Gemini Live expects client audio as 16-bit signed PCM, 16 kHz, mono. */
16
19
  const GEMINI_INPUT_SAMPLE_RATE = 16000;
@@ -18,6 +21,18 @@ const GEMINI_INPUT_SAMPLE_RATE = 16000;
18
21
  const GEMINI_INPUT_AUDIO_MIME_TYPE = 'audio/pcm;rate=16000';
19
22
  /** Gemini Live emits model audio as 16-bit signed PCM, 24 kHz, mono. */
20
23
  const GEMINI_OUTPUT_SAMPLE_RATE = 24000;
24
+ // ── Legacy video-capability fallback ───────────────────────────────────────────
25
+ //
26
+ // Video capability and its rate ceiling are per-model data, minted from the provider's profile
27
+ // table into the session config. A mint from a server that predates those fields sends neither,
28
+ // and this client must still negotiate video for the models that had it — so these two values
29
+ // reproduce the behaviour that shipped before the fields existed, and NOTHING ELSE should read
30
+ // them. They are reachable only against an older server; delete both once no supported server
31
+ // mints a session config without `supportsInboundVideo`.
32
+ /** Model-id prefix that identified a video-capable Live model before the profile carried the flag. */
33
+ const LEGACY_VIDEO_MODEL_PREFIX = 'gemini-3.8-live';
34
+ /** The frame-rate ceiling this client hardcoded before `MaxInboundVideoRate` was minted. */
35
+ const LEGACY_VIDEO_MODEL_RATE = 1;
21
36
  // ── Production playback engine ─────────────────────────────────────────────────
22
37
  /**
23
38
  * Web Audio playout scheduler for Gemini's 24 kHz PCM16 model audio.
@@ -88,13 +103,29 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
88
103
  // ── Transport / audio resources ────────────────────────────────────────────
89
104
  this.session = null;
90
105
  this.micStream = null;
106
+ this.cameraStream = null;
91
107
  this.micCapture = null;
108
+ this.cameraCapture = null;
92
109
  this.playback = null;
110
+ this.firstVideoSendTimestamp = 0;
111
+ this.lastVideoSendTimestamp = 0;
112
+ this.videoFramesSent = 0;
113
+ this.resumptionHandle = null;
114
+ this.lastConnectArgs = null;
115
+ // ── Model capability & profile state ───────────────────────────────────────
116
+ this.idleSignal = 'turnComplete';
117
+ this.supportsScheduling = true;
118
+ this.supportsBlocking = true;
119
+ this.interactionInProgress = false;
120
+ this.toolBatchBarrier = new RealtimeToolBatchBarrier();
121
+ this.assistantSafetyBackstopTimer = null;
93
122
  // ── Response state machine ─────────────────────────────────────────────────
94
123
  /** Accumulates the in-flight assistant transcript across delta frames. */
95
124
  this.pendingAssistantText = '';
96
125
  /** Accumulates the in-flight user transcription across delta frames. */
97
126
  this.pendingUserText = '';
127
+ /** Accumulates in-flight thought text deltas until finalized on turn completion. */
128
+ this.pendingThoughtText = '';
98
129
  /** True while a model turn is in flight; gates (queues) client-triggered sends. */
99
130
  this.responseActive = false;
100
131
  /** The kind of the turn currently in flight; stamped at send time, reset on turnComplete. */
@@ -124,6 +155,34 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
124
155
  */
125
156
  this.currentState = 'closed';
126
157
  }
158
+ static { GeminiRealtimeClient_1 = this; }
159
+ static { this.ASSISTANT_SAFETY_BACKSTOP_MS = 15000; }
160
+ /** Returns the latest session resumption handle reported by the server, if any. */
161
+ get ResumptionHandle() {
162
+ return this.resumptionHandle;
163
+ }
164
+ /** Returns the count of video frames successfully sent over the established video track. */
165
+ get VideoFramesSent() {
166
+ return this.videoFramesSent;
167
+ }
168
+ /**
169
+ * Returns the cumulative active video duration in seconds across sent video frames.
170
+ * Represents the wall-clock span between first and last sent frames (span-not-sum) for stream telemetry.
171
+ * Provider-reported ImageTokens remains the authoritative financial billing basis.
172
+ */
173
+ get VideoSeconds() {
174
+ if (this.firstVideoSendTimestamp === 0 || this.lastVideoSendTimestamp === 0) {
175
+ return 0;
176
+ }
177
+ return Math.max(1, Math.round((this.lastVideoSendTimestamp - this.firstVideoSendTimestamp) / 1000) + 1);
178
+ }
179
+ /**
180
+ * Whether the active model session enforces asynchronous non-blocking tool execution.
181
+ * Derived from model tooling capability (!supportsBlocking), separated from the idle signal (Reviewer Item 19).
182
+ */
183
+ get isNonBlocking() {
184
+ return !this.supportsBlocking;
185
+ }
127
186
  // ── BaseRealtimeClient: connection lifecycle ───────────────────────────────
128
187
  /**
129
188
  * Opens the client-direct Gemini Live session: creates the playout engine, connects with
@@ -131,21 +190,70 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
131
190
  * values the server LOCKED into the token, so tampering is ignored by the API), then wires
132
191
  * the mic-capture worklet. Reports `'listening'` once audio is flowing.
133
192
  */
134
- async Connect(config, micStream) {
193
+ /**
194
+ * Opens the client-direct Gemini Live session: creates the playout engine, connects with
195
+ * the ephemeral token + the server-built `SessionConfig` (`{ model, config }` — the same
196
+ * values the server LOCKED into the token, so tampering is ignored by the API), negotiates
197
+ * tracks, then wires the mic-capture worklet and optional video capture.
198
+ * Reports `'listening'` once audio is flowing.
199
+ */
200
+ async Connect(config, micStream, cameraStream) {
135
201
  this.micStream = micStream;
202
+ this.cameraStream = cameraStream ?? null;
203
+ this.clearSafetyBackstop();
204
+ this.toolBatchBarrier.Clear();
205
+ this.firstVideoSendTimestamp = 0;
206
+ this.lastVideoSendTimestamp = 0;
207
+ this.videoFramesSent = 0;
136
208
  this.setState('connecting');
137
- const { model, liveConfig } = this.parseSessionConfig(config);
209
+ const { model, liveConfig, idleSignal, supportsScheduling, supportsBlocking, supportsInboundVideo, maxInboundVideoRate, requestedTracks } = this.parseSessionConfig(config);
210
+ this.idleSignal = idleSignal;
211
+ this.supportsScheduling = supportsScheduling;
212
+ this.supportsBlocking = supportsBlocking;
213
+ // Negotiate tracks. Video capability and its frame-rate ceiling are PER-MODEL DATA, minted
214
+ // from the provider's profile table (GeminiLiveModelProfile.SupportsInboundVideo /
215
+ // .MaxInboundVideoRate) and carried in the session config. Deriving either from the model
216
+ // id would put a second answer to the same question in a second place: the two agreed only
217
+ // because the model names happened to line up, and the next model to break that pattern
218
+ // would diverge silently.
219
+ const isVideoModel = supportsInboundVideo ?? model.toLowerCase().startsWith(LEGACY_VIDEO_MODEL_PREFIX);
220
+ const supportedTracks = [
221
+ { Modality: 'audio', Direction: 'inbound' },
222
+ { Modality: 'audio', Direction: 'outbound' },
223
+ ];
224
+ if (isVideoModel) {
225
+ supportedTracks.push({
226
+ Modality: 'video',
227
+ Direction: 'inbound',
228
+ Encoding: 'image/jpeg',
229
+ // The model's own ceiling. ResolveRequestedTracks takes the more restrictive of
230
+ // this and what the session requested, so this is what bounds the live track.
231
+ Rate: maxInboundVideoRate ?? LEGACY_VIDEO_MODEL_RATE,
232
+ UsageBasis: ['tokens', 'frames'],
233
+ RequiresConsent: true,
234
+ });
235
+ }
236
+ this.negotiateTracks(requestedTracks, supportedTracks);
138
237
  this.playback = this.createPlayback();
139
- this.session = await this.connectLiveSession({
238
+ const connectArgs = {
140
239
  Model: model,
141
240
  Config: liveConfig,
142
241
  EphemeralToken: config.EphemeralToken,
143
242
  OnMessage: (message) => this.handleServerMessage(message),
144
243
  OnError: (event) => this.handleTransportError(event),
145
- OnClose: () => this.handleTransportClose(),
146
- });
244
+ OnClose: (event) => this.handleTransportClose(event),
245
+ };
246
+ this.lastConnectArgs = connectArgs;
247
+ this.session = await this.connectLiveSession(connectArgs);
147
248
  this.setState('connected');
148
249
  this.micCapture = await this.createMicCapture(micStream, (base64Pcm16) => this.sendMicChunk(base64Pcm16));
250
+ // Start camera capture if inbound video is established and cameraStream provided
251
+ if (this.cameraStream && this.IsTrackEstablished('video', 'inbound')) {
252
+ this.cameraCapture = createStreamFrameCapture(this.cameraStream, {
253
+ Rate: 1,
254
+ OnFrame: (frame) => this.SendVideoFrame(frame.data, frame.mimeType),
255
+ });
256
+ }
149
257
  // Audio-activity capability (base obligation #9): agent side taps the playout
150
258
  // engine's master gain; user side meters the mic stream. Null-safe — test fakes /
151
259
  // no-WebAudio environments simply leave the session un-metered.
@@ -154,18 +262,28 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
154
262
  this.setState('listening');
155
263
  }
156
264
  /**
157
- * Tears down the session, mic capture, mic tracks, and playout engine, resets the response
158
- * state machine, and emits a final `'closed'` (unless already `'error'`). Safe to call
159
- * more than once.
265
+ * Tears down the session, mic capture, mic tracks, camera capture, and playout engine,
266
+ * resets the response state machine, and emits a final `'closed'` (unless already `'error'`).
267
+ * Safe to call more than once.
160
268
  */
161
269
  async Disconnect() {
162
270
  this.closeAudioMeters();
271
+ this.clearSafetyBackstop();
272
+ this.toolBatchBarrier.Clear();
163
273
  this.micStream?.getTracks().forEach((track) => track.stop());
164
274
  this.micStream = null;
275
+ this.cameraCapture?.Stop();
276
+ this.cameraCapture = null;
277
+ this.cameraStream?.getTracks().forEach((track) => track.stop());
278
+ this.cameraStream = null;
165
279
  this.micCapture?.Stop();
166
280
  this.micCapture = null;
167
281
  this.playback?.Close();
168
282
  this.playback = null;
283
+ this.resumptionHandle = null;
284
+ this.firstVideoSendTimestamp = 0;
285
+ this.lastVideoSendTimestamp = 0;
286
+ this.videoFramesSent = 0;
169
287
  if (this.session) {
170
288
  try {
171
289
  this.session.close();
@@ -202,6 +320,40 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
202
320
  this.CancelActiveResponse();
203
321
  this.enqueueOrRun(() => this.sendTriggeringUserTurn(text, 'normal', true));
204
322
  }
323
+ /**
324
+ * Streams one base64 image frame over the established inbound video track.
325
+ *
326
+ * Enforces a 750ms minimum inter-frame spacing to serve as a backstop with deliberate jitter
327
+ * headroom for 1 fps (1000ms) pacers (such as `ChannelInboundVideoBridge`'s `setInterval`,
328
+ * `frameCapture`, and channel-level gates like `OnScreencastFrame`'s 1000ms pacer). Upstream
329
+ * cadence generators and channel gates are the primary enforcers of the nominal 1 fps ceiling,
330
+ * while this 750ms gate absorbs event loop and async dispatch jitter without dropping intended
331
+ * 1Hz frames, while preventing any unpaced callers from bursting above 1.33 fps.
332
+ *
333
+ * If inbound video is not established, returns `false` without error or frame sends (fallback).
334
+ *
335
+ * @returns `true` if the frame was dispatched to the session; `false` if dropped (throttled
336
+ * or track unestablished).
337
+ */
338
+ SendVideoFrame(base64Image, mimeType = 'image/jpeg') {
339
+ if (!this.IsTrackEstablished('video', 'inbound')) {
340
+ return false;
341
+ }
342
+ const now = Date.now();
343
+ if (this.lastVideoSendTimestamp > 0 && now - this.lastVideoSendTimestamp < 750) {
344
+ return false; // Throttled: 750ms jitter headroom backstop for upstream 1 fps pacers (Reviewer Items 25, 30, 33)
345
+ }
346
+ this.lastVideoSendTimestamp = now;
347
+ if (this.firstVideoSendTimestamp === 0) {
348
+ this.firstVideoSendTimestamp = now;
349
+ }
350
+ this.videoFramesSent++;
351
+ // Reviewer Item 31: Send `video` alone — do not populate sibling `media` slot to avoid duplicate bytes & billing
352
+ this.session?.sendRealtimeInput({
353
+ video: { data: base64Image, mimeType },
354
+ });
355
+ return true;
356
+ }
205
357
  /**
206
358
  * @inheritdoc
207
359
  *
@@ -218,11 +370,13 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
218
370
  if (!this.session) {
219
371
  return;
220
372
  }
221
- if (!this.responseActive && !this.IsAudioPlaying) {
373
+ if (!this.responseActive && !this.IsAudioPlaying && !this.interactionInProgress) {
222
374
  return; // nothing active — no-op by contract
223
375
  }
376
+ this.clearSafetyBackstop();
224
377
  this.playback?.Flush();
225
378
  this.responseActive = false;
379
+ this.interactionInProgress = false;
226
380
  this.activeResponseKind = 'normal';
227
381
  this.flushQueuedSends();
228
382
  if (this.currentState === 'speaking') {
@@ -281,7 +435,12 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
281
435
  return;
282
436
  }
283
437
  const name = this.pendingToolCallNames.get(callID) ?? '';
284
- this.enqueueOrRun(() => this.sendToolResponseTurn(callID, name, outputJson));
438
+ if (this.isNonBlocking) {
439
+ this.sendToolResponseTurn(callID, name, outputJson);
440
+ }
441
+ else {
442
+ this.enqueueOrRun(() => this.sendToolResponseTurn(callID, name, outputJson));
443
+ }
285
444
  }
286
445
  /**
287
446
  * Mutes / unmutes by toggling the mic tracks' `enabled` flag: the capture pipeline stays
@@ -297,7 +456,12 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
297
456
  }
298
457
  /** @inheritdoc */
299
458
  get IsBusy() {
300
- return this.responseActive;
459
+ if (this.isNonBlocking) {
460
+ return (this.responseActive ||
461
+ this.interactionInProgress ||
462
+ !this.toolBatchBarrier.IsEmpty);
463
+ }
464
+ return this.responseActive || this.interactionInProgress;
301
465
  }
302
466
  /**
303
467
  * @inheritdoc
@@ -354,21 +518,74 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
354
518
  const model = typeof sessionConfig['model'] === 'string' ? sessionConfig['model'] : config.Model;
355
519
  const raw = sessionConfig['config'];
356
520
  const liveConfig = raw !== null && typeof raw === 'object' && !Array.isArray(raw) ? raw : {};
357
- return { model, liveConfig };
521
+ const rawIdle = sessionConfig['idleSignal'];
522
+ const idleSignal = rawIdle === 'interactionStatus' ? 'interactionStatus' : 'turnComplete';
523
+ const supportsScheduling = sessionConfig['supportsScheduling'] !== false;
524
+ const supportsBlocking = sessionConfig['supportsBlocking'] !== false;
525
+ // Per-model video legality, minted from the provider's profile table. Left undefined by a
526
+ // mint that predates these fields — see LEGACY_VIDEO_MODEL_PREFIX at the call site.
527
+ const supportsInboundVideo = typeof sessionConfig['supportsInboundVideo'] === 'boolean' ? sessionConfig['supportsInboundVideo'] : undefined;
528
+ const rawMaxVideoRate = sessionConfig['maxInboundVideoRate'];
529
+ const maxInboundVideoRate = typeof rawMaxVideoRate === 'number' && rawMaxVideoRate > 0 ? rawMaxVideoRate : undefined;
530
+ const rawRequestedTracks = sessionConfig['requestedTracks'];
531
+ let requestedTracks = undefined;
532
+ if (Array.isArray(rawRequestedTracks)) {
533
+ const list = [];
534
+ for (const item of rawRequestedTracks) {
535
+ if (item !== null && typeof item === 'object' && !Array.isArray(item)) {
536
+ const modality = typeof item['Modality'] === 'string' ? item['Modality'] : undefined;
537
+ const direction = item['Direction'];
538
+ if (modality && (direction === 'inbound' || direction === 'outbound')) {
539
+ list.push({
540
+ Modality: modality,
541
+ Direction: direction,
542
+ Encoding: typeof item['Encoding'] === 'string' ? item['Encoding'] : undefined,
543
+ Rate: typeof item['Rate'] === 'number' ? item['Rate'] : undefined,
544
+ RequiresConsent: typeof item['RequiresConsent'] === 'boolean' ? item['RequiresConsent'] : undefined,
545
+ });
546
+ }
547
+ }
548
+ }
549
+ requestedTracks = list;
550
+ }
551
+ return { model, liveConfig, idleSignal, supportsScheduling, supportsBlocking, supportsInboundVideo, maxInboundVideoRate, requestedTracks };
358
552
  }
359
- /** Streams one base64 PCM16 mic chunk to the model (no-op once the session is gone). */
553
+ /** Streams one base64 PCM16 mic chunk to the model (no-op once the session is gone, closed, or in error). */
360
554
  sendMicChunk(base64Pcm16) {
361
- this.session?.sendRealtimeInput({
362
- audio: { data: base64Pcm16, mimeType: GEMINI_INPUT_AUDIO_MIME_TYPE },
363
- });
555
+ if (!this.session || this.currentState === 'closed' || this.currentState === 'error') {
556
+ return;
557
+ }
558
+ try {
559
+ this.session.sendRealtimeInput({
560
+ audio: { data: base64Pcm16, mimeType: GEMINI_INPUT_AUDIO_MIME_TYPE },
561
+ });
562
+ }
563
+ catch (err) {
564
+ RealtimeDiagLog(`[GeminiRealtimeClient] sendMicChunk failed: ${err instanceof Error ? err.message : String(err)}`);
565
+ }
364
566
  }
365
567
  /** Surfaces a fatal websocket error and marks the session unusable. */
366
568
  handleTransportError(event) {
367
- this.emitError({ Message: `Gemini Live transport error: ${event.message || 'unknown'}`, Fatal: true });
569
+ const detail = event.message || (event.error instanceof Error ? event.error.message : String(event.error ?? 'unknown'));
570
+ RealtimeDiagLog(`[GeminiRealtimeClient] Transport error: ${detail}`);
571
+ this.emitError({ Message: `Gemini Live transport error: ${detail}`, Fatal: true });
368
572
  this.setState('error');
369
573
  }
370
574
  /** Reflects a provider-side close (unless the session already ended in error). */
371
- handleTransportClose() {
575
+ handleTransportClose(event) {
576
+ const code = event?.code;
577
+ const reason = event?.reason;
578
+ const wasClean = event?.wasClean;
579
+ RealtimeDiagLog(`[GeminiRealtimeClient] Transport closed: code=${code} reason=${reason} wasClean=${wasClean}`);
580
+ const isAbnormal = (code !== undefined && code !== 0 && code !== 1000 && code !== 1005) || (wasClean === false && code !== 1000 && code !== 0 && code !== 1005 && code !== undefined);
581
+ if (isAbnormal && this.currentState !== 'error') {
582
+ this.emitError({
583
+ Message: `Gemini Live connection closed (${code}): ${reason || 'unexpected disconnect'}`,
584
+ Fatal: true,
585
+ });
586
+ this.setState('error');
587
+ return;
588
+ }
372
589
  if (this.currentState !== 'error' && this.currentState !== 'closed') {
373
590
  this.setState('closed');
374
591
  }
@@ -379,6 +596,22 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
379
596
  * handlers, including `usageMetadata` → {@link emitUsage}.
380
597
  */
381
598
  handleServerMessage(message) {
599
+ this.checkInteractionStatus(message);
600
+ // Session continuity: track resumption token updates (F7)
601
+ if (message.sessionResumptionUpdate) {
602
+ if (message.sessionResumptionUpdate.resumable === false) {
603
+ this.resumptionHandle = null;
604
+ }
605
+ else if (message.sessionResumptionUpdate.newHandle) {
606
+ this.resumptionHandle = message.sessionResumptionUpdate.newHandle;
607
+ }
608
+ }
609
+ // Server approaching timeout / abort: reconnect seamlessly using resumption handle (F7)
610
+ if (message.goAway) {
611
+ if (this.resumptionHandle) {
612
+ void this.resumeSession(this.resumptionHandle);
613
+ }
614
+ }
382
615
  if (message.serverContent) {
383
616
  this.handleServerContent(message.serverContent);
384
617
  }
@@ -389,6 +622,68 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
389
622
  this.handleUsageMetadata(message.usageMetadata);
390
623
  }
391
624
  }
625
+ /**
626
+ * Resumes the live session using a previously captured session resumption handle (F7).
627
+ */
628
+ async resumeSession(handle) {
629
+ if (!this.lastConnectArgs) {
630
+ return;
631
+ }
632
+ try {
633
+ const reconnectArgs = {
634
+ ...this.lastConnectArgs,
635
+ Config: {
636
+ ...this.lastConnectArgs.Config,
637
+ sessionResumption: { handle },
638
+ },
639
+ };
640
+ const oldSession = this.session;
641
+ const newSession = await this.connectLiveSession(reconnectArgs);
642
+ this.session = newSession;
643
+ this.lastConnectArgs = reconnectArgs;
644
+ try {
645
+ oldSession?.close();
646
+ }
647
+ catch {
648
+ // The old socket has already been replaced by newSession, so a close failure on it cannot affect the new session
649
+ }
650
+ }
651
+ catch (err) {
652
+ RealtimeDiagLog(`[GeminiRealtimeClient] Session resumption failed: ${err instanceof Error ? err.message : String(err)}`);
653
+ }
654
+ }
655
+ /**
656
+ * Inspects inbound frames for the untyped `interaction_status` / `interactionStatus` wire field
657
+ * documented for Gemini Live Extended Thinking (V3). Narrowed via null-safe object/string helpers.
658
+ */
659
+ checkInteractionStatus(message) {
660
+ const msgObj = GeminiRealtimeClient_1.readObject(message);
661
+ const contentObj = GeminiRealtimeClient_1.readObject(message.serverContent);
662
+ const rawStatus = GeminiRealtimeClient_1.readString(msgObj?.['interaction_status']) ??
663
+ GeminiRealtimeClient_1.readString(msgObj?.['interactionStatus']) ??
664
+ GeminiRealtimeClient_1.readString(contentObj?.['interaction_status']) ??
665
+ GeminiRealtimeClient_1.readString(contentObj?.['interactionStatus']);
666
+ if (!rawStatus) {
667
+ return;
668
+ }
669
+ const status = rawStatus.toUpperCase();
670
+ if (status === 'IN_PROGRESS') {
671
+ this.interactionInProgress = true;
672
+ this.responseActive = true;
673
+ if (this.idleSignal === 'interactionStatus') {
674
+ this.scheduleSafetyBackstop();
675
+ }
676
+ }
677
+ else if (status === 'IDLE') {
678
+ this.clearSafetyBackstop();
679
+ if (this.idleSignal === 'interactionStatus') {
680
+ this.handleIdleTerminal();
681
+ }
682
+ else {
683
+ this.interactionInProgress = false;
684
+ }
685
+ }
686
+ }
392
687
  /**
393
688
  * Emits a usage update from a server message's `usageMetadata`.
394
689
  *
@@ -400,9 +695,53 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
400
695
  * the server-bridged `GeminiRealtime` driver forwards the same payload to `IRealtimeSession.OnUsage`.
401
696
  */
402
697
  handleUsageMetadata(usageMetadata) {
698
+ let inputDetails;
699
+ if (usageMetadata.promptTokensDetails && Array.isArray(usageMetadata.promptTokensDetails)) {
700
+ for (const detail of usageMetadata.promptTokensDetails) {
701
+ if (typeof detail.tokenCount === 'number') {
702
+ inputDetails = inputDetails ?? {};
703
+ const mod = String(detail.modality ?? '').toUpperCase();
704
+ if (mod === 'AUDIO') {
705
+ inputDetails.AudioTokens = (inputDetails.AudioTokens ?? 0) + detail.tokenCount;
706
+ }
707
+ else if (mod === 'TEXT') {
708
+ inputDetails.TextTokens = (inputDetails.TextTokens ?? 0) + detail.tokenCount;
709
+ }
710
+ else if (mod === 'IMAGE') {
711
+ inputDetails.ImageTokens = (inputDetails.ImageTokens ?? 0) + detail.tokenCount;
712
+ }
713
+ }
714
+ }
715
+ }
716
+ /**
717
+ * Cost Attribution Note (F6 & Reviewer Item 29):
718
+ * Inbound video frames are sent as individual JPEG images (V5) and billed on the video pricing tier
719
+ * ($0.002 / min, or $1.00 / 1M tokens). The Gemini Live API reports token consumption via
720
+ * usageMetadata.promptTokensDetails partitioned into AUDIO, TEXT, and IMAGE (where video frame tokens
721
+ * are accounted under IMAGE).
722
+ *
723
+ * We expose two candidate cost and telemetry signals:
724
+ * 1. Authoritative Vendor Signal: InputTokenDetails.ImageTokens from promptTokensDetails represents
725
+ * the actual token consumption billed by the Google inference provider.
726
+ * 2. Track-Level Video Telemetry: VideoFrames (cumulative frames sent) and VideoSeconds (cumulative
727
+ * active video duration) client-side counters provide fine-grained telemetry and rate attribution.
728
+ *
729
+ * Logging both signals enables operational drift detection: divergence between client-sent VideoFrames
730
+ * and provider-received ImageTokens immediately surfaces frame drops or network throttling in production.
731
+ */
732
+ if (this.videoFramesSent > 0) {
733
+ inputDetails = inputDetails ?? {};
734
+ inputDetails.VideoFrames = this.videoFramesSent;
735
+ inputDetails.VideoSeconds = this.VideoSeconds;
736
+ }
403
737
  this.emitUsage({
404
738
  InputTokens: typeof usageMetadata.promptTokenCount === 'number' ? usageMetadata.promptTokenCount : undefined,
405
739
  OutputTokens: typeof usageMetadata.responseTokenCount === 'number' ? usageMetadata.responseTokenCount : undefined,
740
+ ...(inputDetails ? { InputTokenDetails: inputDetails } : {}),
741
+ ...(this.videoFramesSent > 0 ? {
742
+ VideoFrames: this.videoFramesSent,
743
+ VideoSeconds: this.VideoSeconds,
744
+ } : {}),
406
745
  Raw: usageMetadata,
407
746
  });
408
747
  }
@@ -413,6 +752,7 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
413
752
  }
414
753
  if (content.modelTurn) {
415
754
  this.handleModelAudio(content.modelTurn);
755
+ this.handleModelThoughts(content.modelTurn);
416
756
  }
417
757
  if (content.inputTranscription) {
418
758
  this.handleUserTranscription(content.inputTranscription);
@@ -420,10 +760,25 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
420
760
  if (content.outputTranscription) {
421
761
  this.handleAssistantTranscription(content.outputTranscription);
422
762
  }
763
+ if (content.generationComplete) {
764
+ this.handleGenerationComplete();
765
+ }
423
766
  if (content.turnComplete) {
424
767
  this.handleTurnComplete();
425
768
  }
426
769
  }
770
+ /**
771
+ * generationComplete: indicates the model has finished generating all tokens for the turn.
772
+ * Playout may still be active (the delay between generationComplete and turnComplete).
773
+ *
774
+ * Per Reviewer Item 20: Draining the queue happens on turnComplete or true IDLE, not prematurely
775
+ * on generationComplete. We set responseActive = false so busy state reflects token completion.
776
+ */
777
+ handleGenerationComplete() {
778
+ if (this.idleSignal === 'turnComplete') {
779
+ this.responseActive = false;
780
+ }
781
+ }
427
782
  /**
428
783
  * Barge-in: the provider stopped generating because the user spoke. Flush every scheduled
429
784
  * playout source (per the Live API contract, `interrupted` is the signal to empty the
@@ -434,6 +789,7 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
434
789
  */
435
790
  handleInterruption() {
436
791
  this.playback?.Flush();
792
+ this.finalizeThoughtTranscript();
437
793
  this.emitInterruption();
438
794
  this.setState('listening');
439
795
  }
@@ -443,6 +799,9 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
443
799
  return;
444
800
  }
445
801
  for (const part of modelTurn.parts) {
802
+ if (part.thought) {
803
+ continue; // Thoughts are reasoning summaries, never spoken audio
804
+ }
446
805
  const data = part.inlineData?.data;
447
806
  if (data) {
448
807
  this.markGenerationStarted();
@@ -450,6 +809,28 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
450
809
  }
451
810
  }
452
811
  }
812
+ /**
813
+ * Extracts thought parts (`part.thought === true`) from model turns and emits them
814
+ * as narration transcript deltas (`Kind: 'narration'`). Thought summaries are reasoning
815
+ * notes, never synthesized as assistant speech.
816
+ */
817
+ handleModelThoughts(modelTurn) {
818
+ if (!modelTurn.parts) {
819
+ return;
820
+ }
821
+ for (const part of modelTurn.parts) {
822
+ if (part.thought && part.text) {
823
+ this.pendingThoughtText += part.text;
824
+ this.emitTranscript({
825
+ Role: 'Assistant',
826
+ Text: part.text,
827
+ IsFinal: false,
828
+ Kind: 'narration',
829
+ IsThought: true,
830
+ });
831
+ }
832
+ }
833
+ }
453
834
  /**
454
835
  * User transcription: each frame's `text` is an incremental DELTA (emitted with
455
836
  * `IsFinal: false`); the accumulated turn text is finalized on the `finished` flag — or,
@@ -487,39 +868,78 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
487
868
  }
488
869
  /**
489
870
  * Surfaces the model's tool calls to the host and caches each callID→name for
490
- * {@link SendToolResult}. Two deliberate behaviors mirror the OpenAI driver: (1) the client
491
- * silently leaves `'speaking'` (no emission) so a host-rendered busy indicator isn't
492
- * clobbered by this turn's trailing frames; (2) `responseActive` is CLEARED — the model has
493
- * yielded the floor pending the result, so a queued tool result can never deadlock waiting
494
- * for a `turnComplete` that may not arrive until after the result is sent.
871
+ * {@link SendToolResult}.
872
+ *
873
+ * In synchronous BLOCKING mode (e.g. 3.1 preview), the model yields the floor pending
874
+ * the tool result: `responseActive` is cleared so the tool response can take the floor,
875
+ * and the client silently leaves 'speaking'.
876
+ *
877
+ * In asynchronous NON_BLOCKING mode (Extended Thinking and 3.8 default), the model keeps
878
+ * generating and reasoning in the background. We must NOT set `responseActive = false`,
879
+ * as the model is not idle and still generating. The tool batch barrier tracks the call.
495
880
  */
496
881
  handleToolCallFrame(functionCalls) {
497
882
  if (!functionCalls || functionCalls.length === 0) {
498
883
  return;
499
884
  }
500
- if (this.currentState === 'speaking') {
501
- this.currentState = 'connected';
885
+ if (!this.isNonBlocking) {
886
+ if (this.currentState === 'speaking') {
887
+ this.currentState = 'connected';
888
+ }
889
+ this.responseActive = false;
502
890
  }
503
- this.responseActive = false;
504
891
  for (const call of functionCalls) {
505
892
  const callID = call.id ?? '';
506
893
  const toolName = call.name ?? '';
507
894
  this.pendingToolCallNames.set(callID, toolName);
895
+ this.toolBatchBarrier.TrackPendingCall(callID, () => {
896
+ this.handleToolBatchTimeout();
897
+ });
508
898
  this.emitToolCall({ CallID: callID, ToolName: toolName, ArgumentsJson: JSON.stringify(call.args ?? {}) });
509
899
  }
510
900
  }
511
901
  /**
512
- * Turn boundary: finalize any un-finished assistant transcript, release the busy lock,
513
- * reset the response kind to `'normal'`, drain queued sends (stopping at the first one
514
- * that starts a new turn), and return the floor to the user.
902
+ * Turn boundary: finalize any un-finished assistant transcript.
903
+ * Under 'turnComplete' idle signal, release the busy lock, reset response kind,
904
+ * drain queued sends, and return the floor to the user.
905
+ * Under 'interactionStatus' (Extended Thinking), turnComplete does NOT indicate idle:
906
+ * background reasoning or async tool calls may still be in flight, so the busy lock
907
+ * and queued sends remain held until the true IDLE signal lands.
515
908
  */
516
909
  handleTurnComplete() {
517
910
  this.finalizeAssistantTranscript();
911
+ this.finalizeThoughtTranscript();
912
+ if (this.idleSignal === 'turnComplete') {
913
+ this.responseActive = false;
914
+ this.activeResponseKind = 'normal';
915
+ this.openClientTurn = false; // the completed generation consumed any open client content
916
+ this.flushQueuedSends();
917
+ if (this.currentState === 'speaking') {
918
+ this.setState('listening');
919
+ }
920
+ }
921
+ else {
922
+ // Extended Thinking: turn complete within active interaction — re-arm liveness guard
923
+ this.scheduleSafetyBackstop();
924
+ }
925
+ }
926
+ /**
927
+ * Reached when the server emits true IDLE (interactionStatus) or the safety backstop fires.
928
+ * Releases busy state, commits any deferred open client turns without cutting off generation,
929
+ * drains queued sends, and returns the floor.
930
+ */
931
+ handleIdleTerminal() {
932
+ this.clearSafetyBackstop();
933
+ this.interactionInProgress = false;
518
934
  this.responseActive = false;
519
935
  this.activeResponseKind = 'normal';
520
- this.openClientTurn = false; // the completed generation consumed any open client content
936
+ this.finalizeThoughtTranscript();
937
+ if (this.openClientTurn) {
938
+ this.session?.sendClientContent({ turnComplete: true });
939
+ this.openClientTurn = false;
940
+ }
521
941
  this.flushQueuedSends();
522
- if (this.currentState === 'speaking') {
942
+ if (this.currentState === 'speaking' && !this.IsAudioPlaying) {
523
943
  this.setState('listening');
524
944
  }
525
945
  }
@@ -531,6 +951,10 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
531
951
  markGenerationStarted() {
532
952
  this.finalizeUserTranscript();
533
953
  this.responseActive = true;
954
+ if (this.idleSignal === 'interactionStatus') {
955
+ this.interactionInProgress = true;
956
+ this.scheduleSafetyBackstop();
957
+ }
534
958
  if (this.currentState !== 'speaking') {
535
959
  this.setState('speaking');
536
960
  }
@@ -551,6 +975,14 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
551
975
  this.emitTranscript({ Role: 'Assistant', Text: text, IsFinal: true, Kind: this.activeResponseKind });
552
976
  }
553
977
  }
978
+ /** Emits the accumulated thought turn as final (if non-empty) with Kind: 'narration' and IsThought: true. */
979
+ finalizeThoughtTranscript() {
980
+ const text = this.pendingThoughtText;
981
+ this.pendingThoughtText = '';
982
+ if (text.trim().length > 0) {
983
+ this.emitTranscript({ Role: 'Assistant', Text: text, IsFinal: true, Kind: 'narration', IsThought: true });
984
+ }
985
+ }
554
986
  // ── Collision-safe send machinery ──────────────────────────────────────────
555
987
  /**
556
988
  * Runs a send immediately when no turn is in flight; otherwise queues it for the next
@@ -595,6 +1027,10 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
595
1027
  // `openClientTurn` is left as-is — the model `turnComplete` that follows clears it.
596
1028
  session.sendRealtimeInput({ text });
597
1029
  this.responseActive = true;
1030
+ if (this.idleSignal === 'interactionStatus') {
1031
+ this.interactionInProgress = true;
1032
+ this.scheduleSafetyBackstop();
1033
+ }
598
1034
  this.activeResponseKind = kind;
599
1035
  if (emitSpeaking) {
600
1036
  this.setState('speaking');
@@ -603,28 +1039,68 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
603
1039
  /**
604
1040
  * Sends the tool response (Gemini continues the turn with it) and marks the model busy.
605
1041
  *
606
- * When context notes have left a client content turn open ({@link openClientTurn}), the
607
- * tool response alone does NOT start generation — the Live API holds for more client input
608
- * until the turn is committed. The empty-turn commit (the SDK-documented
609
- * `sendClientContent({ turnComplete: true })` form) releases generation so the model
610
- * speaks the result immediately, matching the OpenAI driver's explicit `response.create`.
1042
+ * In BLOCKING mode (legacy 3.1), if context notes have left a client content turn open
1043
+ * ({@link openClientTurn}), the tool response alone does NOT start generation — the Live API
1044
+ * holds for more client input until the turn is committed. The empty-turn commit
1045
+ * (`sendClientContent({ turnComplete: true })`) releases generation.
1046
+ *
1047
+ * In NON_BLOCKING mode (Extended Thinking and 3.8 default), setting `turnComplete: true`
1048
+ * unconditionally interrupts active model generation mid-sentence. We must NOT commit
1049
+ * openClientTurn here; it is deferred until true IDLE (or a subsequent user turn commit).
611
1050
  */
1051
+ /**
1052
+ * Extracts scheduling hints (`__mj_scheduling` or legacy `scheduling`) from the tool output,
1053
+ * strips both keys so they do not leak into the model's response payload, resolves the scheduling
1054
+ * directive accepting both 'INTERRUPT' and 'INTERRUPTED', and warns on unrecognized values or
1055
+ * unsupported models (delegates normalization to Core `ExtractToolSchedulingHint`).
1056
+ */
1057
+ resolveFunctionScheduling(parsed, toolName) {
1058
+ const hint = ExtractToolSchedulingHint(parsed, toolName, 'GeminiRealtimeClient');
1059
+ if (!hint) {
1060
+ return undefined;
1061
+ }
1062
+ const schedStr = hint === 'silent' ? 'SILENT' : hint === 'whenIdle' ? 'WHEN_IDLE' : 'INTERRUPT';
1063
+ if (!this.supportsScheduling) {
1064
+ console.warn(`[GeminiRealtimeClient] Dropping scheduling hint "${schedStr}" for tool "${toolName}": ` +
1065
+ `function scheduling is only supported on gemini-3.8-live.`);
1066
+ return undefined;
1067
+ }
1068
+ switch (hint) {
1069
+ case 'silent':
1070
+ return FunctionResponseScheduling.SILENT;
1071
+ case 'whenIdle':
1072
+ return FunctionResponseScheduling.WHEN_IDLE;
1073
+ case 'interrupt':
1074
+ return FunctionResponseScheduling.INTERRUPT;
1075
+ }
1076
+ }
612
1077
  sendToolResponseTurn(callID, name, outputJson) {
613
1078
  const session = this.session;
614
1079
  if (!session) {
615
1080
  return;
616
1081
  }
1082
+ this.toolBatchBarrier.RecordResult(callID);
1083
+ const parsed = this.parseToolOutput(outputJson);
1084
+ const sched = this.resolveFunctionScheduling(parsed, name);
1085
+ const functionResponse = {
1086
+ id: callID,
1087
+ name,
1088
+ response: parsed,
1089
+ ...(sched ? { scheduling: sched } : {}),
1090
+ };
617
1091
  session.sendToolResponse({
618
- functionResponses: [{ id: callID, name, response: this.parseToolOutput(outputJson) }],
1092
+ functionResponses: [functionResponse],
619
1093
  });
620
- if (this.openClientTurn) {
621
- session.sendClientContent({ turnComplete: true });
622
- this.openClientTurn = false;
1094
+ if (!this.isNonBlocking) {
1095
+ if (this.openClientTurn) {
1096
+ session.sendClientContent({ turnComplete: true });
1097
+ this.openClientTurn = false;
1098
+ }
1099
+ this.responseActive = true;
1100
+ this.activeResponseKind = 'normal';
1101
+ this.setState('speaking');
623
1102
  }
624
1103
  this.pendingToolCallNames.delete(callID);
625
- this.responseActive = true;
626
- this.activeResponseKind = 'normal';
627
- this.setState('speaking');
628
1104
  }
629
1105
  /**
630
1106
  * Parses a JSON-stringified tool result into the structured object Gemini's
@@ -644,23 +1120,60 @@ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient
644
1120
  }
645
1121
  }
646
1122
  // ── Helpers ────────────────────────────────────────────────────────────────
1123
+ scheduleSafetyBackstop(delayMs = GeminiRealtimeClient_1.ASSISTANT_SAFETY_BACKSTOP_MS) {
1124
+ this.clearSafetyBackstop();
1125
+ this.assistantSafetyBackstopTimer = setTimeout(() => {
1126
+ this.assistantSafetyBackstopTimer = null;
1127
+ this.handleSafetyBackstopTrigger();
1128
+ }, delayMs);
1129
+ }
1130
+ clearSafetyBackstop() {
1131
+ if (this.assistantSafetyBackstopTimer) {
1132
+ clearTimeout(this.assistantSafetyBackstopTimer);
1133
+ this.assistantSafetyBackstopTimer = null;
1134
+ }
1135
+ }
1136
+ handleSafetyBackstopTrigger() {
1137
+ this.handleIdleTerminal();
1138
+ }
1139
+ handleToolBatchTimeout() {
1140
+ if (this.idleSignal === 'turnComplete') {
1141
+ this.flushQueuedSends();
1142
+ }
1143
+ }
647
1144
  /** Resets the per-session response state machine (used on Disconnect). */
648
1145
  resetResponseState() {
649
1146
  this.pendingAssistantText = '';
650
1147
  this.pendingUserText = '';
1148
+ this.pendingThoughtText = '';
651
1149
  this.responseActive = false;
1150
+ this.interactionInProgress = false;
652
1151
  this.activeResponseKind = 'normal';
653
1152
  this.openClientTurn = false;
654
1153
  this.queuedSends = [];
655
1154
  this.pendingToolCallNames.clear();
1155
+ this.toolBatchBarrier.Clear();
1156
+ this.clearSafetyBackstop();
656
1157
  }
657
1158
  /** Updates the client's own state view and emits the change to the host. */
658
1159
  setState(state) {
659
1160
  this.currentState = state;
660
1161
  this.emitStateChange(state);
661
1162
  }
1163
+ static readObject(value) {
1164
+ return value !== null && typeof value === 'object' && !Array.isArray(value)
1165
+ ? value
1166
+ : undefined;
1167
+ }
1168
+ static readString(value) {
1169
+ if (typeof value !== 'string') {
1170
+ return undefined;
1171
+ }
1172
+ const trimmed = value.trim();
1173
+ return trimmed.length > 0 ? trimmed : undefined;
1174
+ }
662
1175
  };
663
- GeminiRealtimeClient = __decorate([
1176
+ GeminiRealtimeClient = GeminiRealtimeClient_1 = __decorate([
664
1177
  RegisterClass(BaseRealtimeClient, 'gemini')
665
1178
  ], GeminiRealtimeClient);
666
1179
  export { GeminiRealtimeClient };