@memberjunction/ai-realtime-client 0.0.1 → 5.42.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/README.md +148 -28
  2. package/dist/audio/audioMeter.d.ts +101 -0
  3. package/dist/audio/audioMeter.d.ts.map +1 -0
  4. package/dist/audio/audioMeter.js +193 -0
  5. package/dist/audio/audioMeter.js.map +1 -0
  6. package/dist/audio/micCapture.d.ts +26 -0
  7. package/dist/audio/micCapture.d.ts.map +1 -0
  8. package/dist/audio/micCapture.js +69 -0
  9. package/dist/audio/micCapture.js.map +1 -0
  10. package/dist/audio/pcmPlayback.d.ts +73 -0
  11. package/dist/audio/pcmPlayback.d.ts.map +1 -0
  12. package/dist/audio/pcmPlayback.js +78 -0
  13. package/dist/audio/pcmPlayback.js.map +1 -0
  14. package/dist/audio/pcmUtils.d.ts +17 -0
  15. package/dist/audio/pcmUtils.d.ts.map +1 -0
  16. package/dist/audio/pcmUtils.js +46 -0
  17. package/dist/audio/pcmUtils.js.map +1 -0
  18. package/dist/drivers/assemblyAIRealtimeClient.d.ts +384 -0
  19. package/dist/drivers/assemblyAIRealtimeClient.d.ts.map +1 -0
  20. package/dist/drivers/assemblyAIRealtimeClient.js +732 -0
  21. package/dist/drivers/assemblyAIRealtimeClient.js.map +1 -0
  22. package/dist/drivers/elevenLabsRealtimeClient.d.ts +362 -0
  23. package/dist/drivers/elevenLabsRealtimeClient.d.ts.map +1 -0
  24. package/dist/drivers/elevenLabsRealtimeClient.js +686 -0
  25. package/dist/drivers/elevenLabsRealtimeClient.js.map +1 -0
  26. package/dist/drivers/geminiRealtimeClient.d.ts +406 -0
  27. package/dist/drivers/geminiRealtimeClient.d.ts.map +1 -0
  28. package/dist/drivers/geminiRealtimeClient.js +675 -0
  29. package/dist/drivers/geminiRealtimeClient.js.map +1 -0
  30. package/dist/drivers/openAIRealtimeClient.d.ts +381 -0
  31. package/dist/drivers/openAIRealtimeClient.d.ts.map +1 -0
  32. package/dist/drivers/openAIRealtimeClient.js +602 -0
  33. package/dist/drivers/openAIRealtimeClient.js.map +1 -0
  34. package/dist/drivers/xaiRealtimeClient.d.ts +430 -0
  35. package/dist/drivers/xaiRealtimeClient.d.ts.map +1 -0
  36. package/dist/drivers/xaiRealtimeClient.js +676 -0
  37. package/dist/drivers/xaiRealtimeClient.js.map +1 -0
  38. package/dist/generic/baseRealtimeClient.d.ts +401 -0
  39. package/dist/generic/baseRealtimeClient.d.ts.map +1 -0
  40. package/dist/generic/baseRealtimeClient.js +226 -0
  41. package/dist/generic/baseRealtimeClient.js.map +1 -0
  42. package/dist/index.d.ts +11 -0
  43. package/dist/index.d.ts.map +1 -0
  44. package/dist/index.js +11 -0
  45. package/dist/index.js.map +1 -0
  46. package/package.json +28 -7
@@ -0,0 +1,675 @@
1
+ var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
2
+ var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
3
+ if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
4
+ else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
5
+ return c > 3 && r && Object.defineProperty(target, key, r), r;
6
+ };
7
+ import { RegisterClass } from '@memberjunction/global';
8
+ import { GoogleGenAI, } from '@google/genai';
9
+ import { BaseRealtimeClient } from '../generic/baseRealtimeClient.js';
10
+ import { base64ToArrayBuffer } from '../audio/pcmUtils.js';
11
+ import { RealtimePcmPlayback } from '../audio/pcmPlayback.js';
12
+ import { RealtimeAudioMeter } from '../audio/audioMeter.js';
13
+ import { createPcmMicCapture } from '../audio/micCapture.js';
14
+ // ── Audio constants (Gemini Live wire formats) ─────────────────────────────────
15
+ /** Gemini Live expects client audio as 16-bit signed PCM, 16 kHz, mono. */
16
+ const GEMINI_INPUT_SAMPLE_RATE = 16000;
17
+ /** MIME type stamped on every streamed mic chunk. */
18
+ const GEMINI_INPUT_AUDIO_MIME_TYPE = 'audio/pcm;rate=16000';
19
+ /** Gemini Live emits model audio as 16-bit signed PCM, 24 kHz, mono. */
20
+ const GEMINI_OUTPUT_SAMPLE_RATE = 24000;
21
+ // ── Production playback engine ─────────────────────────────────────────────────
22
+ /**
23
+ * Web Audio playout scheduler for Gemini's 24 kHz PCM16 model audio.
24
+ *
25
+ * A thin specialization of the shared {@link RealtimePcmPlayback} (playhead-clock scheduling,
26
+ * gapless playout, honest `IsPlaying`) fixed at Gemini Live's 24 kHz output rate. Kept as a
27
+ * named export for back-compat with existing consumers.
28
+ */
29
+ export class GeminiPcmPlayback extends RealtimePcmPlayback {
30
+ constructor() {
31
+ super(GEMINI_OUTPUT_SAMPLE_RATE);
32
+ }
33
+ }
34
+ // ── The driver ─────────────────────────────────────────────────────────────────
35
+ /**
36
+ * Google Gemini implementation of {@link BaseRealtimeClient}: a **browser-direct** Gemini Live
37
+ * websocket session authenticated with the server-minted ephemeral auth token (`v1alpha`
38
+ * `auth_tokens` mechanism — the token is passed as the SDK `apiKey`).
39
+ *
40
+ * Registered with the ClassFactory under the key `'gemini'` — the `Provider` string the
41
+ * server's `GeminiRealtime` driver stamps on its `ClientRealtimeSessionConfig` — so hosts
42
+ * resolve it without referencing this class directly.
43
+ *
44
+ * Owns ALL Gemini wire concerns (the behavioral twin of {@link OpenAIRealtimeClient}, adapted
45
+ * to Gemini's client-owned audio plane — there is no WebRTC here):
46
+ * - **Audio in**: mic PCM16 @ 16 kHz captured via an inline-Blob `AudioWorklet`
47
+ * ({@link createMicCapture} seam) and streamed with `sendRealtimeInput`.
48
+ * - **Audio out**: model PCM16 @ 24 kHz chunks scheduled through {@link GeminiPcmPlayback}
49
+ * ({@link createPlayback} seam); {@link IsAudioPlaying} is computed from the playout clock.
50
+ * - **Event translation** (Gemini → contract): `inputTranscription` → User deltas/finals,
51
+ * `outputTranscription` → Assistant deltas/finals (accumulated like the OpenAI driver),
52
+ * `toolCall.functionCalls` → {@link OnToolCall} (callID→name cached for
53
+ * {@link SendToolResult}), `interrupted` → playback flush + `'listening'`,
54
+ * `turnComplete` → busy cleared + queued sends flushed, `usageMetadata` → {@link OnUsage}
55
+ * (per-turn prompt/response token deltas — see {@link handleUsageMetadata}).
56
+ * - **Busy mapping**: Gemini has no `response.created` frame, so `IsBusy` is set EAGERLY when
57
+ * this client triggers a response (text / narration / tool result) and on the first model
58
+ * output of a turn (audio part or output-transcription delta); cleared on `turnComplete`,
59
+ * and on a `toolCall` frame (the model has yielded the floor pending the tool result — so a
60
+ * slow `turnComplete` can never deadlock the queued result).
61
+ * - **Collision safety**: ANY `sendClientContent` interrupts in-flight Gemini generation (per
62
+ * the Live API contract), so text / narration / context-note / tool-result sends issued
63
+ * while a turn is in flight are queued and flushed in order on `turnComplete` (the flush
64
+ * stops at the first send that starts a new response).
65
+ * - **Open-turn commit**: context notes ride as `turnComplete: false` client content, which
66
+ * tells the Live API MORE INPUT IS COMING — the server holds ALL generation (including the
67
+ * normally-automatic continuation after a tool response) until a `turnComplete: true`
68
+ * commit. {@link SendToolResult} therefore follows `sendToolResponse` with an empty-turn
69
+ * commit whenever a note left the turn open, so the model speaks the result immediately
70
+ * (the behavioral equivalent of the OpenAI driver's explicit `response.create`).
71
+ - **Triggering turns ride realtime text**: typed text ({@link SendText}) and narration
72
+ * triggers ({@link RequestSpokenUpdate}) are sent via `sendRealtimeInput({ text })` — the
73
+ * Live API's documented in-conversation text path. Native-audio Live models treat
74
+ * `sendClientContent` as initial-history seeding only: a mid-call `turnComplete: true`
75
+ * client turn appends to history WITHOUT starting generation (the model stays silent until
76
+ * the user's next spoken turn), while realtime text triggers an immediate response on every
77
+ * model generation. Context notes stay on `sendClientContent` (`turnComplete: false`) — the
78
+ * silent history-append is exactly the contract they want.
79
+ * - **Narration tagging**: {@link RequestSpokenUpdate} has no per-response-instructions
80
+ * equivalent on Gemini, so it is emulated as a realtime-text user turn carrying the
81
+ * instructions; the response kind is stamped `'narration'` at send time (sends ARE
82
+ * the turn triggers on Gemini, unlike OpenAI where `response.created` confirms) and reset to
83
+ * `'normal'` when the turn completes.
84
+ */
85
+ let GeminiRealtimeClient = class GeminiRealtimeClient extends BaseRealtimeClient {
86
+ constructor() {
87
+ super(...arguments);
88
+ // ── Transport / audio resources ────────────────────────────────────────────
89
+ this.session = null;
90
+ this.micStream = null;
91
+ this.micCapture = null;
92
+ this.playback = null;
93
+ // ── Response state machine ─────────────────────────────────────────────────
94
+ /** Accumulates the in-flight assistant transcript across delta frames. */
95
+ this.pendingAssistantText = '';
96
+ /** Accumulates the in-flight user transcription across delta frames. */
97
+ this.pendingUserText = '';
98
+ /** True while a model turn is in flight; gates (queues) client-triggered sends. */
99
+ this.responseActive = false;
100
+ /** The kind of the turn currently in flight; stamped at send time, reset on turnComplete. */
101
+ this.activeResponseKind = 'normal';
102
+ /** Sends deferred while a turn is in flight; drained in order on turnComplete. */
103
+ this.queuedSends = [];
104
+ /**
105
+ * Maps each pending tool call's `CallID` to its function name: Gemini's `sendToolResponse`
106
+ * requires the function name, which the {@link SendToolResult} contract does not carry.
107
+ */
108
+ this.pendingToolCallNames = new Map();
109
+ /**
110
+ * True while client content sent with `turnComplete: false` (context notes) has not yet
111
+ * been committed. Per the Live API, an open client turn tells the server MORE INPUT IS
112
+ * COMING — clientContent-driven generation (notably the normally-automatic continuation
113
+ * after a tool response) holds until a `turnComplete: true` arrives. Set by
114
+ * {@link SendContextNote}; cleared by the empty-turn commit in
115
+ * {@link sendToolResponseTurn} and on a model `turnComplete` (the generation consumed the
116
+ * open content). Realtime-text triggers ({@link sendTriggeringUserTurn}) do NOT commit
117
+ * client content and leave this flag untouched.
118
+ */
119
+ this.openClientTurn = false;
120
+ /**
121
+ * The client's own view of the session state — mirrors what was last emitted, EXCEPT after
122
+ * a tool call: the host typically shows its own busy indicator then, so the client silently
123
+ * leaves `'speaking'` (no emission) until the result reply's first output re-asserts it.
124
+ */
125
+ this.currentState = 'closed';
126
+ }
127
+ // ── BaseRealtimeClient: connection lifecycle ───────────────────────────────
128
+ /**
129
+ * Opens the client-direct Gemini Live session: creates the playout engine, connects with
130
+ * the ephemeral token + the server-built `SessionConfig` (`{ model, config }` — the same
131
+ * values the server LOCKED into the token, so tampering is ignored by the API), then wires
132
+ * the mic-capture worklet. Reports `'listening'` once audio is flowing.
133
+ */
134
+ async Connect(config, micStream) {
135
+ this.micStream = micStream;
136
+ this.setState('connecting');
137
+ const { model, liveConfig } = this.parseSessionConfig(config);
138
+ this.playback = this.createPlayback();
139
+ this.session = await this.connectLiveSession({
140
+ Model: model,
141
+ Config: liveConfig,
142
+ EphemeralToken: config.EphemeralToken,
143
+ OnMessage: (message) => this.handleServerMessage(message),
144
+ OnError: (event) => this.handleTransportError(event),
145
+ OnClose: () => this.handleTransportClose(),
146
+ });
147
+ this.setState('connected');
148
+ this.micCapture = await this.createMicCapture(micStream, (base64Pcm16) => this.sendMicChunk(base64Pcm16));
149
+ // Audio-activity capability (base obligation #9): agent side taps the playout
150
+ // engine's master gain; user side meters the mic stream. Null-safe — test fakes /
151
+ // no-WebAudio environments simply leave the session un-metered.
152
+ this.attachOutputAudioMeter(this.playback?.CreateMeter?.() ?? null);
153
+ this.attachInputAudioMeter(RealtimeAudioMeter.ForMicStream(micStream));
154
+ this.setState('listening');
155
+ }
156
+ /**
157
+ * Tears down the session, mic capture, mic tracks, and playout engine, resets the response
158
+ * state machine, and emits a final `'closed'` (unless already `'error'`). Safe to call
159
+ * more than once.
160
+ */
161
+ async Disconnect() {
162
+ this.closeAudioMeters();
163
+ this.micStream?.getTracks().forEach((track) => track.stop());
164
+ this.micStream = null;
165
+ this.micCapture?.Stop();
166
+ this.micCapture = null;
167
+ this.playback?.Close();
168
+ this.playback = null;
169
+ if (this.session) {
170
+ try {
171
+ this.session.close();
172
+ }
173
+ catch {
174
+ /* already closing */
175
+ }
176
+ this.session = null;
177
+ }
178
+ this.resetResponseState();
179
+ if (this.currentState !== 'error') {
180
+ this.setState('closed');
181
+ }
182
+ }
183
+ // ── BaseRealtimeClient: outbound actions ──────────────────────────────────
184
+ /**
185
+ * Injects typed text as a realtime-text user turn (`sendRealtimeInput({ text })` —
186
+ * Gemini's "respond now" trigger on every Live model generation, including native-audio
187
+ * models that ignore mid-call `sendClientContent`). No-op when the session is not open.
188
+ *
189
+ * **SendText implies barge-in** (base-contract rule): an active spoken response is
190
+ * cancelled via {@link CancelActiveResponse} first — playback flushed, turn marked
191
+ * inactive, queued sends drained — so the typed turn takes the floor immediately. If a
192
+ * drained queued send (e.g. a pending tool result) starts a new turn, the text queues
193
+ * behind it, preserving the tool-result delivery invariant. On the wire, sending the
194
+ * user turn itself interrupts any residual server-side generation (Gemini Live's
195
+ * any-client-content-interrupts contract), so no explicit cancel frame exists or is
196
+ * needed.
197
+ */
198
+ SendText(text) {
199
+ if (!this.session) {
200
+ return;
201
+ }
202
+ this.CancelActiveResponse();
203
+ this.enqueueOrRun(() => this.sendTriggeringUserTurn(text, 'normal', true));
204
+ }
205
+ /**
206
+ * @inheritdoc
207
+ *
208
+ * Gemini has no explicit cancel frame — the client OWNS the audio plane, so cancelling
209
+ * means: flush the local playout queue ({@link GeminiPcmPlayback}) so speech stops
210
+ * immediately, mark the in-flight turn inactive, and drain queued sends (a queued tool
211
+ * result or context note takes the floor next — tool-result delivery is never dropped
212
+ * by a cancel). Server-side, the next client content sent naturally interrupts any
213
+ * residual generation per the Live API contract. The interrupted turn's accumulated
214
+ * transcript is kept — the provider's trailing frames finalize what WAS spoken. No-op
215
+ * when nothing is active.
216
+ */
217
+ CancelActiveResponse() {
218
+ if (!this.session) {
219
+ return;
220
+ }
221
+ if (!this.responseActive && !this.IsAudioPlaying) {
222
+ return; // nothing active — no-op by contract
223
+ }
224
+ this.playback?.Flush();
225
+ this.responseActive = false;
226
+ this.activeResponseKind = 'normal';
227
+ this.flushQueuedSends();
228
+ if (this.currentState === 'speaking') {
229
+ this.setState('listening');
230
+ }
231
+ }
232
+ /**
233
+ * Injects background context as a user turn with `turnComplete: false` — appended to the
234
+ * conversation WITHOUT starting generation, so the model draws on it the next time it
235
+ * speaks. Gemini Live turns have no system role, so the user role carries it (the host owns
236
+ * prefixing policy). Queued while a turn is in flight because ANY client content interrupts
237
+ * in-flight generation on Gemini (a divergence from the OpenAI driver, which can inject
238
+ * items mid-response safely).
239
+ *
240
+ * SIDE EFFECT: `turnComplete: false` leaves the client content turn OPEN — the server
241
+ * holds generation until a `turnComplete: true` commit arrives. {@link openClientTurn}
242
+ * tracks this so {@link SendToolResult} can commit the turn and unblock the spoken reply.
243
+ */
244
+ SendContextNote(text) {
245
+ if (!this.session) {
246
+ return;
247
+ }
248
+ this.enqueueOrRun(() => {
249
+ this.session?.sendClientContent({ turns: [{ role: 'user', parts: [{ text }] }], turnComplete: false });
250
+ this.openClientTurn = true;
251
+ });
252
+ }
253
+ /**
254
+ * Triggers ONE short spoken update. Gemini has no per-response instructions (OpenAI's
255
+ * `response.create.instructions`), so this is EMULATED: the instructions ride as a
256
+ * realtime-text user turn (`sendRealtimeInput({ text })` — the path that triggers
257
+ * generation on native-audio models, where mid-call `sendClientContent` is inert), and
258
+ * the resulting turn is stamped `Kind: 'narration'` at send time (reset on
259
+ * `turnComplete`) — mirroring the OpenAI driver's narration semantics. Queued behind any
260
+ * in-flight turn so it can never interrupt a pending reply.
261
+ */
262
+ RequestSpokenUpdate(instructions) {
263
+ if (!this.session) {
264
+ return;
265
+ }
266
+ this.enqueueOrRun(() => this.sendTriggeringUserTurn(instructions, 'narration', false));
267
+ }
268
+ /**
269
+ * Feeds an executed tool's result back via `sendToolResponse`, supplying the function name
270
+ * cached from the originating tool call (Gemini requires it; the contract only carries the
271
+ * callID). Sent immediately when idle — Gemini speaks the result as its next turn —
272
+ * otherwise queued until the in-flight turn (e.g. a progress narration) completes.
273
+ *
274
+ * If context notes left a client content turn OPEN (`turnComplete: false`), the tool
275
+ * response is followed by an empty-turn commit (`sendClientContent({ turnComplete: true })`)
276
+ * — without it the server keeps waiting for more client input and NEVER starts the spoken
277
+ * reply (observed live as the model staying silent after a delegated agent's result).
278
+ */
279
+ SendToolResult(callID, outputJson) {
280
+ if (!this.session) {
281
+ return;
282
+ }
283
+ const name = this.pendingToolCallNames.get(callID) ?? '';
284
+ this.enqueueOrRun(() => this.sendToolResponseTurn(callID, name, outputJson));
285
+ }
286
+ /**
287
+ * Mutes / unmutes by toggling the mic tracks' `enabled` flag: the capture pipeline stays
288
+ * up and streams SILENCE while muted (chosen over gating the worklet send so the provider's
289
+ * VAD sees a continuous stream and the un-mute is glitch-free — same policy as the OpenAI
290
+ * client driver).
291
+ */
292
+ SetMuted(muted) {
293
+ const tracks = this.micStream?.getAudioTracks() ?? [];
294
+ for (const track of tracks) {
295
+ track.enabled = !muted;
296
+ }
297
+ }
298
+ /** @inheritdoc */
299
+ get IsBusy() {
300
+ return this.responseActive;
301
+ }
302
+ /**
303
+ * @inheritdoc
304
+ *
305
+ * Computed directly from the playout engine's playhead clock — this client OWNS the output
306
+ * buffer (no WebRTC playback events exist on Gemini), so "audibly playing" is precisely
307
+ * "scheduled audio extends beyond the audio context's current time".
308
+ */
309
+ get IsAudioPlaying() {
310
+ return this.playback?.IsPlaying ?? false;
311
+ }
312
+ // ── Overridable creation seams (tests inject fakes — no network / audio) ──
313
+ /**
314
+ * Creation seam for the Gemini Live session. Production constructs a `GoogleGenAI` client
315
+ * with the **ephemeral token as the API key** on the `v1alpha` API version (the only
316
+ * version that accepts `auth_tokens`), then opens `live.connect` with the server-built
317
+ * model + config. Unit tests override this to return an in-memory fake.
318
+ */
319
+ async connectLiveSession(args) {
320
+ const ai = new GoogleGenAI({ apiKey: args.EphemeralToken, httpOptions: { apiVersion: 'v1alpha' } });
321
+ return ai.live.connect({
322
+ model: args.Model,
323
+ config: args.Config,
324
+ callbacks: {
325
+ onmessage: args.OnMessage,
326
+ onerror: args.OnError,
327
+ onclose: args.OnClose,
328
+ },
329
+ });
330
+ }
331
+ /**
332
+ * Creation seam for the mic-capture pipeline. Production delegates to the shared
333
+ * {@link createPcmMicCapture} (a 16 kHz `AudioContext`, inline-Blob capture worklet, and a
334
+ * zero-gain tail; each worklet block is PCM16-encoded and handed to `onPcmChunk` as base64).
335
+ * Unit tests override this with a no-op fake (and may capture `onPcmChunk` to simulate mic
336
+ * frames).
337
+ */
338
+ async createMicCapture(micStream, onPcmChunk) {
339
+ return createPcmMicCapture(micStream, GEMINI_INPUT_SAMPLE_RATE, onPcmChunk);
340
+ }
341
+ /** Creation seam for the playout engine. Production returns {@link GeminiPcmPlayback}. */
342
+ createPlayback() {
343
+ return new GeminiPcmPlayback();
344
+ }
345
+ // ── Connection internals ───────────────────────────────────────────────────
346
+ /**
347
+ * Extracts the model + Live connect config from the server-minted `SessionConfig`
348
+ * (shaped `{ model, config }` by the server's `GeminiRealtime.CreateClientSession`).
349
+ * Falls back to the top-level `Model` / an empty config if a field is missing — the
350
+ * server locked the real config into the token, so the session still behaves correctly.
351
+ */
352
+ parseSessionConfig(config) {
353
+ const sessionConfig = config.SessionConfig ?? {};
354
+ const model = typeof sessionConfig['model'] === 'string' ? sessionConfig['model'] : config.Model;
355
+ const raw = sessionConfig['config'];
356
+ const liveConfig = raw !== null && typeof raw === 'object' && !Array.isArray(raw) ? raw : {};
357
+ return { model, liveConfig };
358
+ }
359
+ /** Streams one base64 PCM16 mic chunk to the model (no-op once the session is gone). */
360
+ sendMicChunk(base64Pcm16) {
361
+ this.session?.sendRealtimeInput({
362
+ audio: { data: base64Pcm16, mimeType: GEMINI_INPUT_AUDIO_MIME_TYPE },
363
+ });
364
+ }
365
+ /** Surfaces a fatal websocket error and marks the session unusable. */
366
+ handleTransportError(event) {
367
+ this.emitError({ Message: `Gemini Live transport error: ${event.message || 'unknown'}`, Fatal: true });
368
+ this.setState('error');
369
+ }
370
+ /** Reflects a provider-side close (unless the session already ended in error). */
371
+ handleTransportClose() {
372
+ if (this.currentState !== 'error' && this.currentState !== 'closed') {
373
+ this.setState('closed');
374
+ }
375
+ }
376
+ // ── Inbound message translation ────────────────────────────────────────────
377
+ /**
378
+ * Entry point for every inbound {@link LiveServerMessage}; fans out to per-concern
379
+ * handlers, including `usageMetadata` → {@link emitUsage}.
380
+ */
381
+ handleServerMessage(message) {
382
+ if (message.serverContent) {
383
+ this.handleServerContent(message.serverContent);
384
+ }
385
+ if (message.toolCall) {
386
+ this.handleToolCallFrame(message.toolCall.functionCalls);
387
+ }
388
+ if (message.usageMetadata) {
389
+ this.handleUsageMetadata(message.usageMetadata);
390
+ }
391
+ }
392
+ /**
393
+ * Emits a usage update from a server message's `usageMetadata`.
394
+ *
395
+ * **Delta-vs-cumulative (verified against the `@google/genai` SDK types):** `UsageMetadata`
396
+ * documents `promptTokenCount` as "Number of tokens in the prompt" and `responseTokenCount`
397
+ * as "Total number of tokens across all the generated response candidates" — i.e. PER-RESPONSE
398
+ * counts for the turn this message reports, NOT a session-cumulative running total. Each
399
+ * emission is therefore already a delta, matching the `OnUsage` contract — and matching how
400
+ * the server-bridged `GeminiRealtime` driver forwards the same payload to `IRealtimeSession.OnUsage`.
401
+ */
402
+ handleUsageMetadata(usageMetadata) {
403
+ this.emitUsage({
404
+ InputTokens: typeof usageMetadata.promptTokenCount === 'number' ? usageMetadata.promptTokenCount : undefined,
405
+ OutputTokens: typeof usageMetadata.responseTokenCount === 'number' ? usageMetadata.responseTokenCount : undefined,
406
+ Raw: usageMetadata,
407
+ });
408
+ }
409
+ /** Translates one {@link LiveServerContent} frame in provider-documented signal order. */
410
+ handleServerContent(content) {
411
+ if (content.interrupted) {
412
+ this.handleInterruption();
413
+ }
414
+ if (content.modelTurn) {
415
+ this.handleModelAudio(content.modelTurn);
416
+ }
417
+ if (content.inputTranscription) {
418
+ this.handleUserTranscription(content.inputTranscription);
419
+ }
420
+ if (content.outputTranscription) {
421
+ this.handleAssistantTranscription(content.outputTranscription);
422
+ }
423
+ if (content.turnComplete) {
424
+ this.handleTurnComplete();
425
+ }
426
+ }
427
+ /**
428
+ * Barge-in: the provider stopped generating because the user spoke. Flush every scheduled
429
+ * playout source (per the Live API contract, `interrupted` is the signal to empty the
430
+ * client's audio queue), surface the TRUE barge-in to the host (Gemini only emits
431
+ * `interrupted` when user input actually cut off in-flight generation — no extra gating
432
+ * needed), and give the floor back. The interrupted turn's accumulated transcript is
433
+ * kept — the following `turnComplete` finalizes what WAS spoken.
434
+ */
435
+ handleInterruption() {
436
+ this.playback?.Flush();
437
+ this.emitInterruption();
438
+ this.setState('listening');
439
+ }
440
+ /** Decodes inline model-audio parts (base64 PCM16 @ 24 kHz) into the playout queue. */
441
+ handleModelAudio(modelTurn) {
442
+ if (!modelTurn.parts) {
443
+ return;
444
+ }
445
+ for (const part of modelTurn.parts) {
446
+ const data = part.inlineData?.data;
447
+ if (data) {
448
+ this.markGenerationStarted();
449
+ this.playback?.Enqueue(base64ToArrayBuffer(data));
450
+ }
451
+ }
452
+ }
453
+ /**
454
+ * User transcription: each frame's `text` is an incremental DELTA (emitted with
455
+ * `IsFinal: false`); the accumulated turn text is finalized on the `finished` flag — or,
456
+ * because the user's turn is over once the model starts answering, by
457
+ * {@link markGenerationStarted}.
458
+ */
459
+ handleUserTranscription(transcription) {
460
+ if (transcription.text) {
461
+ this.pendingUserText += transcription.text;
462
+ this.emitTranscript({ Role: 'User', Text: transcription.text, IsFinal: false, Kind: 'normal' });
463
+ }
464
+ if (transcription.finished) {
465
+ this.finalizeUserTranscript();
466
+ }
467
+ }
468
+ /**
469
+ * Assistant transcription: deltas accumulate (like the OpenAI driver) and are emitted with
470
+ * the ACTIVE response kind so narration turns are tagged correctly; the turn finalizes on
471
+ * the `finished` flag, with {@link handleTurnComplete} as the fallback.
472
+ */
473
+ handleAssistantTranscription(transcription) {
474
+ this.markGenerationStarted();
475
+ if (transcription.text) {
476
+ this.pendingAssistantText += transcription.text;
477
+ this.emitTranscript({
478
+ Role: 'Assistant',
479
+ Text: transcription.text,
480
+ IsFinal: false,
481
+ Kind: this.activeResponseKind,
482
+ });
483
+ }
484
+ if (transcription.finished) {
485
+ this.finalizeAssistantTranscript();
486
+ }
487
+ }
488
+ /**
489
+ * Surfaces the model's tool calls to the host and caches each callID→name for
490
+ * {@link SendToolResult}. Two deliberate behaviors mirror the OpenAI driver: (1) the client
491
+ * silently leaves `'speaking'` (no emission) so a host-rendered busy indicator isn't
492
+ * clobbered by this turn's trailing frames; (2) `responseActive` is CLEARED — the model has
493
+ * yielded the floor pending the result, so a queued tool result can never deadlock waiting
494
+ * for a `turnComplete` that may not arrive until after the result is sent.
495
+ */
496
+ handleToolCallFrame(functionCalls) {
497
+ if (!functionCalls || functionCalls.length === 0) {
498
+ return;
499
+ }
500
+ if (this.currentState === 'speaking') {
501
+ this.currentState = 'connected';
502
+ }
503
+ this.responseActive = false;
504
+ for (const call of functionCalls) {
505
+ const callID = call.id ?? '';
506
+ const toolName = call.name ?? '';
507
+ this.pendingToolCallNames.set(callID, toolName);
508
+ this.emitToolCall({ CallID: callID, ToolName: toolName, ArgumentsJson: JSON.stringify(call.args ?? {}) });
509
+ }
510
+ }
511
+ /**
512
+ * Turn boundary: finalize any un-finished assistant transcript, release the busy lock,
513
+ * reset the response kind to `'normal'`, drain queued sends (stopping at the first one
514
+ * that starts a new turn), and return the floor to the user.
515
+ */
516
+ handleTurnComplete() {
517
+ this.finalizeAssistantTranscript();
518
+ this.responseActive = false;
519
+ this.activeResponseKind = 'normal';
520
+ this.openClientTurn = false; // the completed generation consumed any open client content
521
+ this.flushQueuedSends();
522
+ if (this.currentState === 'speaking') {
523
+ this.setState('listening');
524
+ }
525
+ }
526
+ /**
527
+ * First model output of a turn (audio part or transcription delta): the user's turn is
528
+ * over (finalize their pending transcript), the model is busy, and the client is audibly /
529
+ * imminently `'speaking'`.
530
+ */
531
+ markGenerationStarted() {
532
+ this.finalizeUserTranscript();
533
+ this.responseActive = true;
534
+ if (this.currentState !== 'speaking') {
535
+ this.setState('speaking');
536
+ }
537
+ }
538
+ /** Emits the accumulated user turn as final (if non-empty) and clears the accumulator. */
539
+ finalizeUserTranscript() {
540
+ const text = this.pendingUserText;
541
+ this.pendingUserText = '';
542
+ if (text.trim().length > 0) {
543
+ this.emitTranscript({ Role: 'User', Text: text, IsFinal: true, Kind: 'normal' });
544
+ }
545
+ }
546
+ /** Emits the accumulated assistant turn as final (if non-empty), tagged with its kind. */
547
+ finalizeAssistantTranscript() {
548
+ const text = this.pendingAssistantText;
549
+ this.pendingAssistantText = '';
550
+ if (text.trim().length > 0) {
551
+ this.emitTranscript({ Role: 'Assistant', Text: text, IsFinal: true, Kind: this.activeResponseKind });
552
+ }
553
+ }
554
+ // ── Collision-safe send machinery ──────────────────────────────────────────
555
+ /**
556
+ * Runs a send immediately when no turn is in flight; otherwise queues it for the next
557
+ * `turnComplete`. This is Gemini's equivalent of the OpenAI driver's
558
+ * queue-behind-active-response rule — stricter here because ANY client content interrupts
559
+ * in-flight generation on the Live API.
560
+ */
561
+ enqueueOrRun(send) {
562
+ if (this.responseActive) {
563
+ this.queuedSends.push(send);
564
+ return;
565
+ }
566
+ send();
567
+ }
568
+ /**
569
+ * Drains queued sends in order on a turn boundary, stopping as soon as one starts a new
570
+ * turn (sets {@link responseActive}) — the rest wait for that turn's completion.
571
+ */
572
+ flushQueuedSends() {
573
+ while (!this.responseActive && this.queuedSends.length > 0) {
574
+ const send = this.queuedSends.shift();
575
+ send?.();
576
+ }
577
+ }
578
+ /**
579
+ * Sends a user turn that TRIGGERS generation (`turnComplete: true`), stamping the upcoming
580
+ * turn's kind at send time and eagerly marking the model busy. `emitSpeaking` mirrors the
581
+ * OpenAI driver: typed-text / tool-result replies reflect `'speaking'` immediately;
582
+ * narration waits for the first model output.
583
+ */
584
+ sendTriggeringUserTurn(text, kind, emitSpeaking) {
585
+ const session = this.session;
586
+ if (!session) {
587
+ return;
588
+ }
589
+ // REALTIME text, not clientContent: native-audio Live models only honor clientContent
590
+ // for history seeding — a mid-call `turnComplete: true` user turn lands in history but
591
+ // does NOT start generation (the model stays silent until the user's next spoken turn
592
+ // commits via VAD). `sendRealtimeInput({ text })` is the documented in-conversation
593
+ // text path and triggers an immediate response on every Live model generation.
594
+ // NOTE: realtime text does NOT commit an open clientContent turn (context notes), so
595
+ // `openClientTurn` is left as-is — the model `turnComplete` that follows clears it.
596
+ session.sendRealtimeInput({ text });
597
+ this.responseActive = true;
598
+ this.activeResponseKind = kind;
599
+ if (emitSpeaking) {
600
+ this.setState('speaking');
601
+ }
602
+ }
603
+ /**
604
+ * Sends the tool response (Gemini continues the turn with it) and marks the model busy.
605
+ *
606
+ * When context notes have left a client content turn open ({@link openClientTurn}), the
607
+ * tool response alone does NOT start generation — the Live API holds for more client input
608
+ * until the turn is committed. The empty-turn commit (the SDK-documented
609
+ * `sendClientContent({ turnComplete: true })` form) releases generation so the model
610
+ * speaks the result immediately, matching the OpenAI driver's explicit `response.create`.
611
+ */
612
+ sendToolResponseTurn(callID, name, outputJson) {
613
+ const session = this.session;
614
+ if (!session) {
615
+ return;
616
+ }
617
+ session.sendToolResponse({
618
+ functionResponses: [{ id: callID, name, response: this.parseToolOutput(outputJson) }],
619
+ });
620
+ if (this.openClientTurn) {
621
+ session.sendClientContent({ turnComplete: true });
622
+ this.openClientTurn = false;
623
+ }
624
+ this.pendingToolCallNames.delete(callID);
625
+ this.responseActive = true;
626
+ this.activeResponseKind = 'normal';
627
+ this.setState('speaking');
628
+ }
629
+ /**
630
+ * Parses a JSON-stringified tool result into the structured object Gemini's
631
+ * function-response slot expects, wrapping non-object output as `{ result: <value> }` so a
632
+ * free-text result still round-trips (same fallback as the server driver).
633
+ */
634
+ parseToolOutput(output) {
635
+ try {
636
+ const parsed = JSON.parse(output);
637
+ if (parsed !== null && typeof parsed === 'object' && !Array.isArray(parsed)) {
638
+ return parsed;
639
+ }
640
+ return { result: parsed };
641
+ }
642
+ catch {
643
+ return { result: output };
644
+ }
645
+ }
646
+ // ── Helpers ────────────────────────────────────────────────────────────────
647
+ /** Resets the per-session response state machine (used on Disconnect). */
648
+ resetResponseState() {
649
+ this.pendingAssistantText = '';
650
+ this.pendingUserText = '';
651
+ this.responseActive = false;
652
+ this.activeResponseKind = 'normal';
653
+ this.openClientTurn = false;
654
+ this.queuedSends = [];
655
+ this.pendingToolCallNames.clear();
656
+ }
657
+ /** Updates the client's own state view and emits the change to the host. */
658
+ setState(state) {
659
+ this.currentState = state;
660
+ this.emitStateChange(state);
661
+ }
662
+ };
663
+ GeminiRealtimeClient = __decorate([
664
+ RegisterClass(BaseRealtimeClient, 'gemini')
665
+ ], GeminiRealtimeClient);
666
+ export { GeminiRealtimeClient };
667
+ /**
668
+ * Tree-shaking prevention: bundlers cannot see that {@link GeminiRealtimeClient} is
669
+ * instantiated dynamically through the ClassFactory, so a consumer must call this no-op
670
+ * to create a static code path that keeps the `@RegisterClass` side effect alive.
671
+ */
672
+ export function LoadGeminiRealtimeClient() {
673
+ // intentional no-op — the static import of this module is the point
674
+ }
675
+ //# sourceMappingURL=geminiRealtimeClient.js.map