@memberjunction/ai-realtime-client 5.47.0 → 5.49.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,11 +5,9 @@ var __decorate = (this && this.__decorate) || function (decorators, target, key,
5
5
  return c > 3 && r && Object.defineProperty(target, key, r), r;
6
6
  };
7
7
  import { RegisterClass } from '@memberjunction/global';
8
+ import { IsTranscriptContinuation } from '@memberjunction/ai';
8
9
  import { BaseRealtimeClient } from '../generic/baseRealtimeClient.js';
9
- import { base64ToArrayBuffer } from '../audio/pcmUtils.js';
10
- import { RealtimePcmPlayback } from '../audio/pcmPlayback.js';
11
- import { RealtimeAudioMeter } from '../audio/audioMeter.js';
12
- import { createPcmMicCapture } from '../audio/micCapture.js';
10
+ import { OpenAIProtocolWebSocketRealtimeClient, } from '../generic/openAIProtocolClient.js';
13
11
  // ── Audio + endpoint constants (xAI Grok Voice wire format) ────────────────────
14
12
  /**
15
13
  * The Grok Voice realtime websocket endpoint. The model is appended as a `?model=` query
@@ -41,42 +39,21 @@ export const XAI_PCM_SAMPLE_RATE = 24000;
41
39
  * matching Grok Voice driver stamps on its `ClientRealtimeSessionConfig` — so hosts resolve it
42
40
  * without referencing this class directly.
43
41
  *
44
- * It is the structural CROSS of the two sibling drivers:
45
- * - Like the AssemblyAI client driver, the transport is a WEBSOCKET and the audio plane is
46
- * CLIENT-OWNED: mic PCM16 streams out as base64 frames via the shared {@link createPcmMicCapture}
47
- * worklet, and the agent's base64 audio chunks feed the shared {@link RealtimePcmPlayback}
48
- * engine ({@link IsAudioPlaying} comes from its playout clock). There is NO SDP handshake, NO
49
- * peer connection, NO remote-audio sink element.
50
- * - Like the OpenAI client driver, the EVENT PROTOCOL is OpenAI-Realtime-API compatible: the
51
- * same client events (session.update, conversation.item.create, response.create,
52
- * response.cancel, input_audio_buffer.append) and the same server events
53
- * (response.output_audio_transcript.delta and .done, response.audio.delta,
54
- * conversation.item.input_audio_transcription.completed,
55
- * response.function_call_arguments.done, response.created, response.done,
56
- * input_audio_buffer.speech_started, error). The response state machine, narration-kind
57
- * tagging, and tool-result queueing are identical to the OpenAI driver.
42
+ * Grok Voice speaks the OpenAI Realtime wire protocol over a websocket with a client-owned PCM
43
+ * audio plane, so nearly everything lives in the shared layers: the protocol brain
44
+ * (`OpenAIProtocolRealtimeClient`) and the websocket+PCM transport
45
+ * ({@link OpenAIProtocolWebSocketRealtimeClient}). This class supplies only the Grok
46
+ * specifics: the model-on-URL endpoint + subprotocol auth, the FIXED 24 kHz audio format,
47
+ * Grok's STREAMED input-transcription behavior, and wire diagnostics.
58
48
  *
59
- * Connect handshake: open the model-on-URL endpoint with the subprotocol auth → on socket open,
60
- * apply the server-authored `config.SessionConfig` via a `session.update` frame → build the
61
- * audio plane at the provider's FIXED 24 kHz PCM16 format → report `'listening'`. The
62
- * `'listening'` gate is the socket-open + session.update applied point (obligation #7): unlike
63
- * AssemblyAI there is no separate `session.ready` confirmation in this protocol, so applying the
64
- * config on open is the readiness boundary.
49
+ * Connect handshake (shared skeleton): open the socket → apply the server-authored
50
+ * `config.SessionConfig` via `session.update` on OPEN (this protocol has no separate readiness
51
+ * ack, so applying the config on open is the readiness boundary — obligation #7) → build the
52
+ * audio plane at 24 kHz → report `'listening'`.
65
53
  */
66
- let xAIRealtimeClient = class xAIRealtimeClient extends BaseRealtimeClient {
54
+ let xAIRealtimeClient = class xAIRealtimeClient extends OpenAIProtocolWebSocketRealtimeClient {
67
55
  constructor() {
68
56
  super(...arguments);
69
- // ── Transport / audio resources ────────────────────────────────────────────
70
- this.socket = null;
71
- this.micStream = null;
72
- this.micCapture = null;
73
- this.playback = null;
74
- /**
75
- * The server-built session config applied verbatim via `session.update` once the socket
76
- * opens. Protected so test subclasses can seed it without a full Connect.
77
- */
78
- this.sessionConfig = null;
79
- // ── Response state machine (mirrors the OpenAI client driver) ──────────────
80
57
  /**
81
58
  * Whether the CURRENT user turn has already emitted a transcription. Grok streams input
82
59
  * transcription as repeated `.completed` events (each the full growing text), so the first emission
@@ -84,264 +61,34 @@ let xAIRealtimeClient = class xAIRealtimeClient extends BaseRealtimeClient {
84
61
  * stack of growing duplicates. Reset on each `input_audio_buffer.speech_started` (new turn).
85
62
  */
86
63
  this.userTurnTranscribed = false;
87
- /** Accumulates the in-flight assistant transcript across delta frames. */
88
- this.pendingAssistantText = '';
89
- /** True while the model has a response in flight; gates narration + queues the tool result. */
90
- this.responseActive = false;
91
- /** Set when a tool result is ready while a response is active; sent on the next response.done. */
92
- this.pendingResultResponse = false;
93
- /**
94
- * Set by {@link RequestSpokenUpdate} just before it sends its `response.create`, and
95
- * CONSUMED by the very next `response.created` frame, which stamps
96
- * {@link activeResponseKind} for that turn.
97
- */
98
- this.pendingNarrationKind = false;
99
64
  /**
100
- * The kind of the response currently in flight. Event ordering: `response.created` →
101
- * transcript deltas → `*_audio_transcript.done` → `response.done`. The transcript-done
102
- * frame therefore arrives while the kind is still set, letting {@link onAssistantDone}
103
- * classify the turn; `response.done` then resets the kind to `'normal'`.
65
+ * Text of the current user turn's latest streamed caption, used to recognize that a post-pause
66
+ * caption CONTINUES the same utterance (Grok re-emits the full accumulated text) rather than
67
+ * starting a new turn. Cleared in {@link onResponseStarted} — once the model replies, the user's
68
+ * turn is over and a later utterance must never merge into it. Mirrors the server session.
104
69
  */
105
- this.activeResponseKind = 'normal';
106
- /**
107
- * The client's own view of the session state — mirrors what {@link emitStateChange} last
108
- * reported, EXCEPT after a tool call is emitted: the host typically shows its own busy
109
- * indicator then, so the client silently leaves `'speaking'` (no emission) to preserve the
110
- * host's indicator until the result reply starts (see {@link handleEvent}).
111
- */
112
- this.currentState = 'closed';
113
- /** True once Disconnect ran — an expected socket close must not surface as fatal. */
114
- this.closedByConsumer = false;
70
+ this.lastUserTranscript = '';
115
71
  }
116
- // ── BaseRealtimeClient: connection lifecycle ───────────────────────────────
117
- /**
118
- * Opens the client-direct Grok Voice websocket: the model-on-URL endpoint with the
119
- * subprotocol auth, the server-authored `config.SessionConfig` applied via `session.update`
120
- * once the socket opens, then the audio plane at the provider's fixed 24 kHz format. Reports
121
- * `'listening'` only after all of that (obligation #7).
122
- */
123
- async Connect(config, micStream) {
124
- this.sessionConfig = config.SessionConfig;
125
- this.micStream = micStream;
126
- this.closedByConsumer = false;
127
- this.setState('connecting');
128
- let openSocket = null;
129
- let failOpen = null;
130
- const opened = new Promise((resolve, reject) => {
131
- openSocket = resolve;
132
- failOpen = reject;
133
- });
72
+ /** @inheritdoc — used in the shared diagnostics + close messages. */
73
+ get providerDebugLabel() {
74
+ return 'xAIRealtimeClient';
75
+ }
76
+ // ── Transport hooks ────────────────────────────────────────────────────────
77
+ /** @inheritdoc — model on the URL, ephemeral secret as the subprotocol. */
78
+ openProviderSocket(config) {
134
79
  const url = `${XAI_REALTIME_WS_URL}?model=${encodeURIComponent(config.Model)}`;
135
80
  const subprotocol = `${XAI_CLIENT_SECRET_SUBPROTOCOL_PREFIX}${config.EphemeralToken}`;
136
- const socket = this.createSocket(url, subprotocol);
137
- this.socket = socket;
138
- socket.onopen = () => openSocket?.();
139
- socket.onmessage = (data) => this.handleSocketMessage(data);
140
- socket.onerror = (message) => {
141
- failOpen?.(new Error(message));
142
- this.handleSocketError(message);
143
- };
144
- socket.onclose = () => {
145
- failOpen?.(new Error('xAI Grok Voice socket closed during connect'));
146
- this.handleSocketClose();
147
- };
148
- await opened;
149
- this.setState('connected');
150
- // The server-authored session config (the SessionConfig pact) is applied as the FIRST
151
- // frame — prompt and tool authority stay server-side (obligation #8). This protocol has
152
- // no separate readiness ack, so applying the config on open IS the readiness boundary.
153
- this.applySessionConfig();
154
- this.playback = this.createPlayback(XAI_PCM_SAMPLE_RATE);
155
- this.micCapture = await this.createMicCapture(micStream, XAI_PCM_SAMPLE_RATE, (base64Pcm16) => this.sendMicChunk(base64Pcm16));
156
- // Audio-activity capability (base obligation #9): agent side taps the playout engine's
157
- // master gain; user side meters the mic stream. Null-safe — test fakes / no-WebAudio
158
- // environments simply leave the session un-metered.
159
- this.attachOutputAudioMeter(this.playback?.CreateMeter?.() ?? null);
160
- this.attachInputAudioMeter(RealtimeAudioMeter.ForMicStream(micStream));
161
- this.setState('listening');
162
- }
163
- /**
164
- * Tears down the socket, mic capture, mic tracks, and playout engine, resets the response
165
- * state machine, and emits a final `'closed'` (unless already `'error'`). Safe to call more
166
- * than once.
167
- */
168
- async Disconnect() {
169
- this.closedByConsumer = true;
170
- this.closeAudioMeters();
171
- this.micStream?.getTracks().forEach((track) => track.stop());
172
- this.micStream = null;
173
- this.micCapture?.Stop();
174
- this.micCapture = null;
175
- this.playback?.Close();
176
- this.playback = null;
177
- if (this.socket) {
178
- try {
179
- this.socket.close();
180
- }
181
- catch {
182
- /* already closing */
183
- }
184
- this.socket = null;
185
- }
186
- this.sessionConfig = null;
187
- this.resetResponseState();
188
- if (this.currentState !== 'error') {
189
- this.setState('closed');
190
- }
81
+ return this.createSocket(url, subprotocol);
191
82
  }
192
- // ── BaseRealtimeClient: outbound actions ──────────────────────────────────
193
- /**
194
- * Injects typed text as a user-role `message` conversation item, then triggers a reply
195
- * through the SAME collision-safe path tool results use ({@link requestResultResponse}).
196
- * No-op when the socket isn't open.
197
- *
198
- * **SendText implies barge-in** (base-contract rule): an active spoken response is cancelled
199
- * via {@link CancelActiveResponse} before the text is injected, so the typed turn takes the
200
- * floor immediately instead of waiting behind stale speech. When nothing is active the cancel
201
- * is a no-op and the reply triggers immediately.
202
- */
203
- SendText(text) {
204
- if (!this.socket) {
205
- return;
206
- }
207
- this.CancelActiveResponse();
208
- this.sendEvent({
209
- type: 'conversation.item.create',
210
- item: {
211
- type: 'message',
212
- role: 'user',
213
- content: [{ type: 'input_text', text }],
214
- },
215
- });
216
- this.requestResultResponse();
83
+ /** @inheritdoc — the Grok Voice wire format is fixed at 24 kHz PCM16 both directions. */
84
+ resolveSampleRate(_config) {
85
+ return XAI_PCM_SAMPLE_RATE;
217
86
  }
218
- /**
219
- * @inheritdoc
220
- *
221
- * Sends `response.cancel` (only when a response is actually in flight) and flushes the
222
- * locally-owned playout queue so already-generated speech stops coming out of the speaker
223
- * immediately. Resets the local response state machine (active flag, narration kind,
224
- * accumulated transcript) but PRESERVES any queued tool-result trigger: delegated work is
225
- * never affected by a floor-control cancel, and the queued trigger still fires on the
226
- * cancelled response's trailing `response.done`. No-op when idle or when the socket is gone.
227
- */
228
- CancelActiveResponse() {
229
- if (!this.socket) {
230
- return;
231
- }
232
- if (!this.responseActive && !this.IsAudioPlaying) {
233
- return; // nothing active — no-op by contract
234
- }
235
- if (this.responseActive) {
236
- this.sendEvent({ type: 'response.cancel' });
237
- this.responseActive = false;
238
- this.pendingNarrationKind = false;
239
- this.activeResponseKind = 'normal';
240
- this.pendingAssistantText = '';
241
- }
242
- // The client OWNS the audio plane (websocket transport) — flush the local playout queue
243
- // so speech stops immediately (obligation #3), no provider clear frame is needed.
244
- this.playback?.Flush();
245
- if (this.currentState === 'speaking') {
246
- this.setState('listening');
247
- }
248
- }
249
- /**
250
- * Injects a system-role context item the model can draw on the next time it speaks, WITHOUT
251
- * forcing a reply. Item creation is always safe mid-response, so it is sent immediately even
252
- * while a reply is in flight.
253
- *
254
- * NOTE: role must be 'system' — the OpenAI-compatible realtime API rejects 'developer' items.
255
- */
256
- SendContextNote(text) {
257
- if (!this.socket) {
258
- return;
259
- }
260
- this.sendEvent({
261
- type: 'conversation.item.create',
262
- item: {
263
- type: 'message',
264
- role: 'system',
265
- content: [{ type: 'input_text', text }],
266
- },
267
- });
268
- }
269
- /**
270
- * Triggers ONE short spoken update with the given instructions. Marks the upcoming response
271
- * as `'narration'` (flag consumed by the next `response.created`) so its transcripts are
272
- * emitted with `Kind: 'narration'` — ephemeral by contract. Sets {@link responseActive}
273
- * eagerly so a tool result landing mid-narration queues instead of colliding.
274
- *
275
- * **Skips when busy** (base-contract collision rule — drivers MUST queue or skip): a
276
- * `response.create` sent while a response is in flight would be rejected/garbled by the
277
- * provider, and narration is disposable by contract, so the update is dropped with a debug
278
- * log rather than queued to come out late and stale. Hosts SHOULD still gate on
279
- * {@link IsBusy} / {@link IsAudioPlaying} for timing quality.
280
- */
281
- RequestSpokenUpdate(instructions) {
282
- if (!this.socket) {
283
- return;
284
- }
285
- if (this.responseActive) {
286
- console.debug('[xAIRealtimeClient] RequestSpokenUpdate skipped — a response is already in flight (narration is disposable).');
287
- return;
288
- }
289
- this.responseActive = true;
290
- this.pendingNarrationKind = true;
291
- this.sendEvent({ type: 'response.create', response: { instructions } });
292
- }
293
- /**
294
- * Sends the tool result back as a `function_call_output` conversation item, then triggers a
295
- * reply — immediately if the model is idle, otherwise queued until the current response
296
- * (e.g. a progress narration) finishes. Without the queueing the result's `response.create`
297
- * would collide with an in-flight narration and be dropped, leaving the model silent when
298
- * delegated work comes back.
299
- */
300
- SendToolResult(callID, outputJson) {
301
- if (!this.socket) {
302
- return;
303
- }
304
- this.sendEvent({
305
- type: 'conversation.item.create',
306
- item: {
307
- type: 'function_call_output',
308
- call_id: callID,
309
- output: outputJson,
310
- },
311
- });
312
- this.requestResultResponse();
313
- }
314
- /**
315
- * Mutes / unmutes by toggling the mic tracks' `enabled` flag: the capture pipeline stays up
316
- * and streams SILENCE while muted (the provider's VAD sees a continuous stream and the
317
- * un-mute is glitch-free — same policy as the other client drivers).
318
- */
319
- SetMuted(muted) {
320
- const tracks = this.micStream?.getAudioTracks() ?? [];
321
- for (const track of tracks) {
322
- track.enabled = !muted;
323
- }
324
- }
325
- /** @inheritdoc */
326
- get IsBusy() {
327
- return this.responseActive;
328
- }
329
- /**
330
- * @inheritdoc
331
- *
332
- * Computed directly from the playout engine's playhead clock — this client OWNS the output
333
- * buffer, so "audibly playing" is precisely "scheduled audio extends beyond the audio
334
- * context's current time".
335
- */
336
- get IsAudioPlaying() {
337
- return this.playback?.IsPlaying ?? false;
338
- }
339
- // ── Overridable creation seams (tests inject fakes — no network / audio) ──
340
87
  /**
341
88
  * Creation seam for the realtime websocket. Production wraps the platform-global `WebSocket`
342
89
  * opened against the model-on-URL endpoint WITH the `xai-client-secret.<token>` subprotocol
343
90
  * (browser auth — no handshake header is possible); unit tests override this to return an
344
- * in-memory fake. Handlers are attached by {@link Connect} AFTER this returns, so the
91
+ * in-memory fake. Handlers are attached by the shared Connect AFTER this returns, so the
345
92
  * implementation must not require them at construction time.
346
93
  */
347
94
  createSocket(url, subprotocol) {
@@ -364,301 +111,85 @@ let xAIRealtimeClient = class xAIRealtimeClient extends BaseRealtimeClient {
364
111
  ws.onclose = () => seam.onclose?.();
365
112
  return seam;
366
113
  }
114
+ // ── Grok behavior overrides ────────────────────────────────────────────────
367
115
  /**
368
- * Creation seam for the mic-capture pipeline at the provider's fixed 24 kHz rate. Production
369
- * delegates to the shared {@link createPcmMicCapture}; unit tests override this with a no-op
370
- * fake (and may capture `onPcmChunk` to simulate mic frames).
371
- */
372
- async createMicCapture(micStream, sampleRate, onPcmChunk) {
373
- return createPcmMicCapture(micStream, sampleRate, onPcmChunk);
374
- }
375
- /**
376
- * Creation seam for the playout engine at the provider's fixed 24 kHz rate. Production
377
- * returns the shared {@link RealtimePcmPlayback}.
378
- */
379
- createPlayback(sampleRate) {
380
- return new RealtimePcmPlayback(sampleRate);
381
- }
382
- // ── Connection internals ───────────────────────────────────────────────────
383
- /**
384
- * Sends the server-controlled session config (instructions + tools) as a `session.update` so
385
- * the co-agent's identity and tool set apply. Skipped when the host supplied no config (e.g.
386
- * it failed to parse the server payload — the host already logged that; sending an EMPTY
387
- * `session.update` would be wrong).
388
- */
389
- applySessionConfig() {
390
- if (!this.sessionConfig || Object.keys(this.sessionConfig).length === 0) {
391
- return;
392
- }
393
- this.sendEvent({ type: 'session.update', session: this.sessionConfig });
394
- }
395
- /** Streams one base64 PCM16 mic chunk as an `input_audio_buffer.append` frame. */
396
- sendMicChunk(base64Pcm16) {
397
- if (this.socket) {
398
- this.sendEvent({ type: 'input_audio_buffer.append', audio: base64Pcm16 });
399
- }
400
- }
401
- /** Surfaces a fatal socket error and marks the session unusable (obligation #6). */
402
- handleSocketError(message) {
403
- if (this.currentState === 'error' || this.currentState === 'closed') {
404
- return;
405
- }
406
- this.emitError({ Message: `xAI Grok Voice realtime transport error: ${message}`, Fatal: true });
407
- this.setState('error');
408
- }
409
- /**
410
- * A socket close the CONSUMER didn't ask for is fatal: the provider hard-closes at token
411
- * expiry and when it ends the session itself, so an unexpected close is how credential /
412
- * session death reaches the host (obligation #6).
413
- */
414
- handleSocketClose() {
415
- if (this.closedByConsumer || this.currentState === 'error' || this.currentState === 'closed') {
416
- return;
417
- }
418
- this.emitError({ Message: 'xAI Grok Voice realtime connection closed unexpectedly', Fatal: true });
419
- this.setState('error');
420
- }
421
- // ── Inbound message translation ────────────────────────────────────────────
422
- /** Parses one raw socket payload; non-JSON frames are ignored. */
423
- handleSocketMessage(data) {
424
- let event;
425
- try {
426
- event = JSON.parse(data);
427
- }
428
- catch {
429
- console.debug('[xAIRealtimeClient] ◀ inbound NON-JSON frame:', String(data).slice(0, 200));
430
- return; // non-JSON frame — ignore
431
- }
432
- if (event === null || typeof event !== 'object') {
433
- return; // valid JSON but not an event object (e.g. "null", a number) — ignore
434
- }
435
- // DIAGNOSTIC: log every inbound event type so a "connected but silent" session is debuggable —
436
- // surfaces error frames + any event whose name diverges from the OpenAI-compatible set (which
437
- // the handleEvent switch would otherwise drop silently). Error frames also log their payload.
438
- console.debug('[xAIRealtimeClient] ◀ inbound:', event.type, event.type === 'error' ? JSON.stringify(event).slice(0, 400) : '');
439
- this.handleEvent(event);
440
- }
441
- /** Dispatches a typed xAI realtime server event to the appropriate behavior. */
442
- handleEvent(event) {
443
- switch (event.type) {
444
- case 'response.output_audio_transcript.delta':
445
- case 'response.audio_transcript.delta':
446
- this.onAssistantDelta(event.delta);
447
- break;
448
- case 'response.output_audio_transcript.done':
449
- case 'response.audio_transcript.done':
450
- this.onAssistantDone(event.transcript);
451
- break;
452
- case 'response.audio.delta':
453
- case 'response.output_audio.delta':
454
- this.onAudioDelta(event.delta);
455
- break;
456
- case 'conversation.item.input_audio_transcription.completed':
457
- this.onUserTranscript(event.transcript);
458
- break;
459
- case 'response.function_call_arguments.done':
460
- this.onToolCallFrame(event);
461
- break;
462
- case 'input_audio_buffer.speech_started':
463
- this.onSpeechStarted();
464
- break;
465
- case 'response.created':
466
- this.responseActive = true;
467
- // Stamp the kind of THIS response: 'narration' only when the flag was set by
468
- // RequestSpokenUpdate immediately before its response.create (consumed here).
469
- this.activeResponseKind = this.pendingNarrationKind ? 'narration' : 'normal';
470
- this.pendingNarrationKind = false;
471
- break;
472
- case 'response.done':
473
- // A turn finished — release the lock and speak any queued tool result so the
474
- // model always voices the answer when delegated work comes back. The
475
- // transcript-done frame for this turn has already arrived (it precedes
476
- // response.done), so it's safe to reset the response kind here.
477
- this.responseActive = false;
478
- this.activeResponseKind = 'normal';
479
- this.emitResponseUsage(event);
480
- this.flushPendingResultResponse();
481
- if (this.currentState === 'speaking') {
482
- this.setState('listening');
483
- }
484
- break;
485
- case 'error':
486
- this.onErrorFrame(event);
487
- break;
488
- default:
489
- // Unhandled event types are expected (the provider emits many); no-op.
490
- break;
491
- }
492
- }
493
- /** Appends an assistant transcript delta, reflects `'speaking'`, and emits the delta. */
494
- onAssistantDelta(delta) {
495
- if (this.currentState !== 'speaking') {
496
- this.setState('speaking');
497
- }
498
- this.pendingAssistantText += delta;
499
- this.emitTranscript({ Role: 'Assistant', Text: delta, IsFinal: false, Kind: this.activeResponseKind });
500
- }
501
- /**
502
- * Finalizes the assistant turn: emits the final transcript tagged with the ACTIVE response
503
- * kind (the transcript-done frame arrives BEFORE `response.done`, so
504
- * {@link activeResponseKind} still reflects this turn), then returns to `'listening'`. Empty
505
- * turns emit nothing.
506
- */
507
- onAssistantDone(transcript) {
508
- const finalText = transcript || this.pendingAssistantText;
509
- this.pendingAssistantText = '';
510
- if (finalText.trim().length > 0) {
511
- this.emitTranscript({ Role: 'Assistant', Text: finalText, IsFinal: true, Kind: this.activeResponseKind });
512
- }
513
- if (this.currentState === 'speaking') {
514
- this.setState('listening');
515
- }
516
- }
517
- /**
518
- * Decodes one base64 PCM16 chunk of the agent's spoken output into the playout queue and
519
- * reflects `'speaking'`. The client OWNS the audio plane on this websocket transport (unlike
520
- * the WebRTC OpenAI driver where the peer connection plays the remote track), so agent audio
521
- * arrives as these deltas and is scheduled by the shared playout engine.
522
- */
523
- onAudioDelta(deltaBase64) {
524
- if (!deltaBase64) {
525
- return;
526
- }
527
- if (this.currentState !== 'speaking') {
528
- this.setState('speaking');
529
- }
530
- this.playback?.Enqueue(base64ToArrayBuffer(deltaBase64));
531
- }
532
- /**
533
- * Emits the user's spoken-input transcription. Grok STREAMS this — repeated
534
- * `input_audio_transcription.completed` events, each carrying the full text so far — unlike OpenAI's
535
- * single final. So the FIRST emission of a turn appends a fresh caption and every later one is flagged
536
- * ReplacesPrevious, collapsing the stream into ONE in-place-updating user bubble. The per-turn flag
537
- * resets on the next `speech_started` ({@link onSpeechStarted}).
116
+ * @inheritdoc
117
+ *
118
+ * Grok STREAMS the input transcription — repeated `input_audio_transcription.completed`
119
+ * events, each carrying the full text so far — unlike OpenAI's single final. So the FIRST
120
+ * emission of a turn appends a fresh caption and every later one is flagged
121
+ * ReplacesPrevious, collapsing the stream into ONE in-place-updating user bubble. The
122
+ * per-turn flag resets on the next `speech_started` ({@link onSpeechStartedFrame}).
538
123
  */
539
- onUserTranscript(transcript) {
124
+ onUserTranscriptFrame(transcript) {
540
125
  if (transcript && transcript.trim().length > 0) {
126
+ // Two ways this caption REPLACES rather than appends: another already landed in this turn,
127
+ // OR it continues the previous utterance (the provider re-emits the whole thing with more
128
+ // words). The second case rescues a mid-sentence pause — the VAD fires speech_started and
129
+ // clears the flag even though the user never stopped, which would otherwise split one
130
+ // spoken thought into several growing bubbles. Mirrors the server session exactly so both
131
+ // topologies collapse the stream identically.
132
+ const replacesPrevious = this.userTurnTranscribed || IsTranscriptContinuation(this.lastUserTranscript, transcript);
541
133
  this.emitTranscript({
542
134
  Role: 'User', Text: transcript, IsFinal: true, Kind: 'normal',
543
- ReplacesPrevious: this.userTurnTranscribed,
135
+ ReplacesPrevious: replacesPrevious,
544
136
  });
545
137
  this.userTurnTranscribed = true;
138
+ this.lastUserTranscript = transcript;
546
139
  }
547
140
  }
548
141
  /**
549
- * Surfaces a completed tool call to the host. Two deliberate behaviors mirror the OpenAI
550
- * driver: (1) the client silently leaves `'speaking'` (NO emission) so a host-rendered busy
551
- * indicator isn't clobbered by this turn's trailing `response.done` / playback frames
552
- * (obligation #1); (2) {@link responseActive} is CLEARED — the model has yielded the floor
553
- * pending the result, so a queued {@link SendToolResult} can never deadlock (obligation #2).
142
+ * @inheritdoc
143
+ *
144
+ * On top of the shared websocket behavior (flush local playout on TRUE barge-in, gated
145
+ * interruption): a new user turn begins, so the streamed-transcription flag resets, and an
146
+ * interrupted response's busy flag is cleared eagerly (Grok cancels its own turn; clearing
147
+ * now lets a queued tool result fire without waiting on the trailing `response.done`).
554
148
  */
555
- onToolCallFrame(call) {
556
- if (this.currentState === 'speaking') {
557
- this.currentState = 'connected';
558
- }
559
- this.responseActive = false;
560
- this.emitToolCall({ CallID: call.call_id, ToolName: call.name, ArgumentsJson: call.arguments });
561
- }
562
149
  /**
563
- * The user started speaking. This is a TRUE barge-in only when it cut off active model output
564
- * (a response in flight or audio audibly playing) — a normal turn while the model is idle is
565
- * NOT an interruption, so the emission is gated (base-contract rule). On a true barge-in the
566
- * client OWNS the audio plane, so it flushes its own playout queue NOW (obligation #3),
567
- * surfaces the interruption, and returns the floor; the provider cancels its own turn and
568
- * emits a terminal `response.done`.
150
+ * @inheritdoc
151
+ *
152
+ * The model has taken the floor, so the user's turn is definitively over — drop the continuation
153
+ * anchor. Without this, a later utterance that happened to open with the same words could be
154
+ * merged into the previous turn instead of appending its own.
569
155
  */
570
- onSpeechStarted() {
571
- // A new user turn begins: the next transcription emission appends a fresh caption, and the rest
572
- // of this turn's streamed transcriptions replace it in place (see onUserTranscript).
156
+ onResponseStarted() {
157
+ this.lastUserTranscript = '';
158
+ }
159
+ onSpeechStartedFrame() {
573
160
  this.userTurnTranscribed = false;
574
- if (!this.responseActive && !this.IsAudioPlaying) {
575
- this.setState('listening');
576
- return;
161
+ if (this.responseActive || this.IsAudioPlaying) {
162
+ this.playback?.Flush();
163
+ this.responseActive = false;
164
+ this.activeResponseKind = 'normal';
165
+ // Floor to the user — drop the queued auto-trigger (see the brain's docstring).
166
+ this.pendingResultResponse = false;
167
+ this.emitInterruption();
577
168
  }
578
- this.playback?.Flush();
579
- this.responseActive = false;
580
- this.activeResponseKind = 'normal';
581
- this.emitInterruption();
582
169
  this.setState('listening');
583
170
  }
584
- /**
585
- * Emits the completed response's usage to the host as a DELTA (the `response.done` usage
586
- * payload covers exactly this response, so it is already incremental — the `OnUsage`
587
- * contract's preferred shape). Frames without a usage payload emit nothing.
588
- */
589
- emitResponseUsage(event) {
590
- const usage = event.response?.usage;
591
- if (!usage) {
592
- return;
593
- }
594
- this.emitUsage({
595
- InputTokens: typeof usage.input_tokens === 'number' ? usage.input_tokens : undefined,
596
- OutputTokens: typeof usage.output_tokens === 'number' ? usage.output_tokens : undefined,
597
- Raw: usage,
598
- });
599
- }
600
- /** Surfaces a provider error frame (non-fatal; the session continues). */
601
- onErrorFrame(event) {
602
- this.emitError({
603
- Message: event.error?.message ?? 'Unknown provider error',
604
- Code: event.error?.code,
605
- Fatal: false,
606
- });
607
- }
608
- // ── Response state machine ─────────────────────────────────────────────────
609
- /**
610
- * Asks the model to speak (a tool result or typed-text reply) — immediately if it's idle,
611
- * otherwise queued until the current response finishes. An immediate trigger also CONSUMES
612
- * any queued trigger debt: every payload item is already in the conversation, so one
613
- * `response.create` voices everything (e.g. typed text barging in over a narration that had
614
- * tool results queued behind it).
615
- */
616
- requestResultResponse() {
617
- if (!this.socket) {
618
- return;
619
- }
620
- if (this.responseActive) {
621
- this.pendingResultResponse = true;
622
- return;
623
- }
624
- this.pendingResultResponse = false;
625
- this.responseActive = true;
626
- this.sendEvent({ type: 'response.create' });
627
- this.setState('speaking');
171
+ /** @inheritdoc — provider-branded transport-error message (kept stable for hosts/logs). */
172
+ formatTransportError(message) {
173
+ return `xAI Grok Voice realtime transport error: ${message}`;
628
174
  }
629
- /** On a turn completing, fire any queued tool-result response so the answer is spoken. */
630
- flushPendingResultResponse() {
631
- if (!this.pendingResultResponse || !this.socket) {
632
- return;
633
- }
634
- this.pendingResultResponse = false;
635
- this.responseActive = true;
636
- this.sendEvent({ type: 'response.create' });
637
- this.setState('speaking');
175
+ /** @inheritdoc — provider-branded unexpected-close message (kept stable for hosts/logs). */
176
+ get unexpectedCloseMessage() {
177
+ return 'xAI Grok Voice realtime connection closed unexpectedly';
638
178
  }
639
- /** Resets the per-session response state machine (used on Disconnect). */
640
- resetResponseState() {
641
- this.pendingAssistantText = '';
642
- this.responseActive = false;
643
- this.pendingResultResponse = false;
644
- this.pendingNarrationKind = false;
645
- this.activeResponseKind = 'normal';
179
+ // ── Wire diagnostics (a silent Grok session is otherwise undebuggable) ────
180
+ /** @inheritdoc — log every inbound event type; error frames include their payload. */
181
+ logInboundEvent(event) {
182
+ console.debug('[xAIRealtimeClient] ◀ inbound:', event.type, event.type === 'error' ? JSON.stringify(event).slice(0, 400) : '');
646
183
  }
647
- // ── Helpers ────────────────────────────────────────────────────────────────
648
- /** Updates the client's own state view and emits the change to the host. */
649
- setState(state) {
650
- this.currentState = state;
651
- this.emitStateChange(state);
184
+ /** @inheritdoc */
185
+ onNonJsonFrame(raw) {
186
+ console.debug('[xAIRealtimeClient] ◀ inbound NON-JSON frame:', String(raw).slice(0, 200));
652
187
  }
653
- /** JSON-serializes + sends a client event over the socket (only when open). */
654
- sendEvent(event) {
655
- // DIAGNOSTIC: log outbound control frames (skip the high-frequency mic audio) so we can see
656
- // session.update / conversation.item.create / response.create actually went out — the other
657
- // half of diagnosing a silent session.
188
+ /** @inheritdoc — log outbound control frames (skip the high-frequency mic audio). */
189
+ logOutboundEvent(event) {
658
190
  if (event.type !== 'input_audio_buffer.append') {
659
191
  console.debug('[xAIRealtimeClient] ▶ outbound:', event.type);
660
192
  }
661
- this.socket?.send(JSON.stringify(event));
662
193
  }
663
194
  };
664
195
  xAIRealtimeClient = __decorate([