@memberjunction/ai-realtime-client 5.48.0 → 5.50.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -0
- package/dist/drivers/huggingFaceRealtimeClient.d.ts +27 -148
- package/dist/drivers/huggingFaceRealtimeClient.d.ts.map +1 -1
- package/dist/drivers/huggingFaceRealtimeClient.js +41 -388
- package/dist/drivers/huggingFaceRealtimeClient.js.map +1 -1
- package/dist/drivers/openAIRealtimeClient.d.ts +28 -270
- package/dist/drivers/openAIRealtimeClient.d.ts.map +1 -1
- package/dist/drivers/openAIRealtimeClient.js +42 -379
- package/dist/drivers/openAIRealtimeClient.js.map +1 -1
- package/dist/drivers/xaiRealtimeClient.d.ts +57 -370
- package/dist/drivers/xaiRealtimeClient.d.ts.map +1 -1
- package/dist/drivers/xaiRealtimeClient.js +85 -554
- package/dist/drivers/xaiRealtimeClient.js.map +1 -1
- package/dist/generic/openAIProtocolClient.d.ts +527 -0
- package/dist/generic/openAIProtocolClient.d.ts.map +1 -0
- package/dist/generic/openAIProtocolClient.js +873 -0
- package/dist/generic/openAIProtocolClient.js.map +1 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/index.js.map +1 -1
- package/package.json +3 -3
|
@@ -5,11 +5,9 @@ var __decorate = (this && this.__decorate) || function (decorators, target, key,
|
|
|
5
5
|
return c > 3 && r && Object.defineProperty(target, key, r), r;
|
|
6
6
|
};
|
|
7
7
|
import { RegisterClass } from '@memberjunction/global';
|
|
8
|
+
import { IsTranscriptContinuation } from '@memberjunction/ai';
|
|
8
9
|
import { BaseRealtimeClient } from '../generic/baseRealtimeClient.js';
|
|
9
|
-
import {
|
|
10
|
-
import { RealtimePcmPlayback } from '../audio/pcmPlayback.js';
|
|
11
|
-
import { RealtimeAudioMeter } from '../audio/audioMeter.js';
|
|
12
|
-
import { createPcmMicCapture } from '../audio/micCapture.js';
|
|
10
|
+
import { OpenAIProtocolWebSocketRealtimeClient, } from '../generic/openAIProtocolClient.js';
|
|
13
11
|
// ── Audio + endpoint constants (xAI Grok Voice wire format) ────────────────────
|
|
14
12
|
/**
|
|
15
13
|
* The Grok Voice realtime websocket endpoint. The model is appended as a `?model=` query
|
|
@@ -41,42 +39,21 @@ export const XAI_PCM_SAMPLE_RATE = 24000;
|
|
|
41
39
|
* matching Grok Voice driver stamps on its `ClientRealtimeSessionConfig` — so hosts resolve it
|
|
42
40
|
* without referencing this class directly.
|
|
43
41
|
*
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
*
|
|
47
|
-
*
|
|
48
|
-
*
|
|
49
|
-
*
|
|
50
|
-
* - Like the OpenAI client driver, the EVENT PROTOCOL is OpenAI-Realtime-API compatible: the
|
|
51
|
-
* same client events (session.update, conversation.item.create, response.create,
|
|
52
|
-
* response.cancel, input_audio_buffer.append) and the same server events
|
|
53
|
-
* (response.output_audio_transcript.delta and .done, response.audio.delta,
|
|
54
|
-
* conversation.item.input_audio_transcription.completed,
|
|
55
|
-
* response.function_call_arguments.done, response.created, response.done,
|
|
56
|
-
* input_audio_buffer.speech_started, error). The response state machine, narration-kind
|
|
57
|
-
* tagging, and tool-result queueing are identical to the OpenAI driver.
|
|
42
|
+
* Grok Voice speaks the OpenAI Realtime wire protocol over a websocket with a client-owned PCM
|
|
43
|
+
* audio plane, so nearly everything lives in the shared layers: the protocol brain
|
|
44
|
+
* (`OpenAIProtocolRealtimeClient`) and the websocket+PCM transport
|
|
45
|
+
* ({@link OpenAIProtocolWebSocketRealtimeClient}). This class supplies only the Grok
|
|
46
|
+
* specifics: the model-on-URL endpoint + subprotocol auth, the FIXED 24 kHz audio format,
|
|
47
|
+
* Grok's STREAMED input-transcription behavior, and wire diagnostics.
|
|
58
48
|
*
|
|
59
|
-
* Connect handshake: open the
|
|
60
|
-
*
|
|
61
|
-
*
|
|
62
|
-
*
|
|
63
|
-
* AssemblyAI there is no separate `session.ready` confirmation in this protocol, so applying the
|
|
64
|
-
* config on open is the readiness boundary.
|
|
49
|
+
* Connect handshake (shared skeleton): open the socket → apply the server-authored
|
|
50
|
+
* `config.SessionConfig` via `session.update` on OPEN (this protocol has no separate readiness
|
|
51
|
+
* ack, so applying the config on open is the readiness boundary — obligation #7) → build the
|
|
52
|
+
* audio plane at 24 kHz → report `'listening'`.
|
|
65
53
|
*/
|
|
66
|
-
let xAIRealtimeClient = class xAIRealtimeClient extends
|
|
54
|
+
let xAIRealtimeClient = class xAIRealtimeClient extends OpenAIProtocolWebSocketRealtimeClient {
|
|
67
55
|
constructor() {
|
|
68
56
|
super(...arguments);
|
|
69
|
-
// ── Transport / audio resources ────────────────────────────────────────────
|
|
70
|
-
this.socket = null;
|
|
71
|
-
this.micStream = null;
|
|
72
|
-
this.micCapture = null;
|
|
73
|
-
this.playback = null;
|
|
74
|
-
/**
|
|
75
|
-
* The server-built session config applied verbatim via `session.update` once the socket
|
|
76
|
-
* opens. Protected so test subclasses can seed it without a full Connect.
|
|
77
|
-
*/
|
|
78
|
-
this.sessionConfig = null;
|
|
79
|
-
// ── Response state machine (mirrors the OpenAI client driver) ──────────────
|
|
80
57
|
/**
|
|
81
58
|
* Whether the CURRENT user turn has already emitted a transcription. Grok streams input
|
|
82
59
|
* transcription as repeated `.completed` events (each the full growing text), so the first emission
|
|
@@ -84,264 +61,34 @@ let xAIRealtimeClient = class xAIRealtimeClient extends BaseRealtimeClient {
|
|
|
84
61
|
* stack of growing duplicates. Reset on each `input_audio_buffer.speech_started` (new turn).
|
|
85
62
|
*/
|
|
86
63
|
this.userTurnTranscribed = false;
|
|
87
|
-
/** Accumulates the in-flight assistant transcript across delta frames. */
|
|
88
|
-
this.pendingAssistantText = '';
|
|
89
|
-
/** True while the model has a response in flight; gates narration + queues the tool result. */
|
|
90
|
-
this.responseActive = false;
|
|
91
|
-
/** Set when a tool result is ready while a response is active; sent on the next response.done. */
|
|
92
|
-
this.pendingResultResponse = false;
|
|
93
|
-
/**
|
|
94
|
-
* Set by {@link RequestSpokenUpdate} just before it sends its `response.create`, and
|
|
95
|
-
* CONSUMED by the very next `response.created` frame, which stamps
|
|
96
|
-
* {@link activeResponseKind} for that turn.
|
|
97
|
-
*/
|
|
98
|
-
this.pendingNarrationKind = false;
|
|
99
64
|
/**
|
|
100
|
-
*
|
|
101
|
-
*
|
|
102
|
-
*
|
|
103
|
-
*
|
|
65
|
+
* Text of the current user turn's latest streamed caption, used to recognize that a post-pause
|
|
66
|
+
* caption CONTINUES the same utterance (Grok re-emits the full accumulated text) rather than
|
|
67
|
+
* starting a new turn. Cleared in {@link onResponseStarted} — once the model replies, the user's
|
|
68
|
+
* turn is over and a later utterance must never merge into it. Mirrors the server session.
|
|
104
69
|
*/
|
|
105
|
-
this.
|
|
106
|
-
/**
|
|
107
|
-
* The client's own view of the session state — mirrors what {@link emitStateChange} last
|
|
108
|
-
* reported, EXCEPT after a tool call is emitted: the host typically shows its own busy
|
|
109
|
-
* indicator then, so the client silently leaves `'speaking'` (no emission) to preserve the
|
|
110
|
-
* host's indicator until the result reply starts (see {@link handleEvent}).
|
|
111
|
-
*/
|
|
112
|
-
this.currentState = 'closed';
|
|
113
|
-
/** True once Disconnect ran — an expected socket close must not surface as fatal. */
|
|
114
|
-
this.closedByConsumer = false;
|
|
70
|
+
this.lastUserTranscript = '';
|
|
115
71
|
}
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
async Connect(config, micStream) {
|
|
124
|
-
this.sessionConfig = config.SessionConfig;
|
|
125
|
-
this.micStream = micStream;
|
|
126
|
-
this.closedByConsumer = false;
|
|
127
|
-
this.setState('connecting');
|
|
128
|
-
let openSocket = null;
|
|
129
|
-
let failOpen = null;
|
|
130
|
-
const opened = new Promise((resolve, reject) => {
|
|
131
|
-
openSocket = resolve;
|
|
132
|
-
failOpen = reject;
|
|
133
|
-
});
|
|
72
|
+
/** @inheritdoc — used in the shared diagnostics + close messages. */
|
|
73
|
+
get providerDebugLabel() {
|
|
74
|
+
return 'xAIRealtimeClient';
|
|
75
|
+
}
|
|
76
|
+
// ── Transport hooks ────────────────────────────────────────────────────────
|
|
77
|
+
/** @inheritdoc — model on the URL, ephemeral secret as the subprotocol. */
|
|
78
|
+
openProviderSocket(config) {
|
|
134
79
|
const url = `${XAI_REALTIME_WS_URL}?model=${encodeURIComponent(config.Model)}`;
|
|
135
80
|
const subprotocol = `${XAI_CLIENT_SECRET_SUBPROTOCOL_PREFIX}${config.EphemeralToken}`;
|
|
136
|
-
|
|
137
|
-
this.socket = socket;
|
|
138
|
-
socket.onopen = () => openSocket?.();
|
|
139
|
-
socket.onmessage = (data) => this.handleSocketMessage(data);
|
|
140
|
-
socket.onerror = (message) => {
|
|
141
|
-
failOpen?.(new Error(message));
|
|
142
|
-
this.handleSocketError(message);
|
|
143
|
-
};
|
|
144
|
-
socket.onclose = () => {
|
|
145
|
-
failOpen?.(new Error('xAI Grok Voice socket closed during connect'));
|
|
146
|
-
this.handleSocketClose();
|
|
147
|
-
};
|
|
148
|
-
await opened;
|
|
149
|
-
this.setState('connected');
|
|
150
|
-
// The server-authored session config (the SessionConfig pact) is applied as the FIRST
|
|
151
|
-
// frame — prompt and tool authority stay server-side (obligation #8). This protocol has
|
|
152
|
-
// no separate readiness ack, so applying the config on open IS the readiness boundary.
|
|
153
|
-
this.applySessionConfig();
|
|
154
|
-
this.playback = this.createPlayback(XAI_PCM_SAMPLE_RATE);
|
|
155
|
-
this.micCapture = await this.createMicCapture(micStream, XAI_PCM_SAMPLE_RATE, (base64Pcm16) => this.sendMicChunk(base64Pcm16));
|
|
156
|
-
// Audio-activity capability (base obligation #9): agent side taps the playout engine's
|
|
157
|
-
// master gain; user side meters the mic stream. Null-safe — test fakes / no-WebAudio
|
|
158
|
-
// environments simply leave the session un-metered.
|
|
159
|
-
this.attachOutputAudioMeter(this.playback?.CreateMeter?.() ?? null);
|
|
160
|
-
this.attachInputAudioMeter(RealtimeAudioMeter.ForMicStream(micStream));
|
|
161
|
-
this.setState('listening');
|
|
162
|
-
}
|
|
163
|
-
/**
|
|
164
|
-
* Tears down the socket, mic capture, mic tracks, and playout engine, resets the response
|
|
165
|
-
* state machine, and emits a final `'closed'` (unless already `'error'`). Safe to call more
|
|
166
|
-
* than once.
|
|
167
|
-
*/
|
|
168
|
-
async Disconnect() {
|
|
169
|
-
this.closedByConsumer = true;
|
|
170
|
-
this.closeAudioMeters();
|
|
171
|
-
this.micStream?.getTracks().forEach((track) => track.stop());
|
|
172
|
-
this.micStream = null;
|
|
173
|
-
this.micCapture?.Stop();
|
|
174
|
-
this.micCapture = null;
|
|
175
|
-
this.playback?.Close();
|
|
176
|
-
this.playback = null;
|
|
177
|
-
if (this.socket) {
|
|
178
|
-
try {
|
|
179
|
-
this.socket.close();
|
|
180
|
-
}
|
|
181
|
-
catch {
|
|
182
|
-
/* already closing */
|
|
183
|
-
}
|
|
184
|
-
this.socket = null;
|
|
185
|
-
}
|
|
186
|
-
this.sessionConfig = null;
|
|
187
|
-
this.resetResponseState();
|
|
188
|
-
if (this.currentState !== 'error') {
|
|
189
|
-
this.setState('closed');
|
|
190
|
-
}
|
|
81
|
+
return this.createSocket(url, subprotocol);
|
|
191
82
|
}
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
* through the SAME collision-safe path tool results use ({@link requestResultResponse}).
|
|
196
|
-
* No-op when the socket isn't open.
|
|
197
|
-
*
|
|
198
|
-
* **SendText implies barge-in** (base-contract rule): an active spoken response is cancelled
|
|
199
|
-
* via {@link CancelActiveResponse} before the text is injected, so the typed turn takes the
|
|
200
|
-
* floor immediately instead of waiting behind stale speech. When nothing is active the cancel
|
|
201
|
-
* is a no-op and the reply triggers immediately.
|
|
202
|
-
*/
|
|
203
|
-
SendText(text) {
|
|
204
|
-
if (!this.socket) {
|
|
205
|
-
return;
|
|
206
|
-
}
|
|
207
|
-
this.CancelActiveResponse();
|
|
208
|
-
this.sendEvent({
|
|
209
|
-
type: 'conversation.item.create',
|
|
210
|
-
item: {
|
|
211
|
-
type: 'message',
|
|
212
|
-
role: 'user',
|
|
213
|
-
content: [{ type: 'input_text', text }],
|
|
214
|
-
},
|
|
215
|
-
});
|
|
216
|
-
this.requestResultResponse();
|
|
83
|
+
/** @inheritdoc — the Grok Voice wire format is fixed at 24 kHz PCM16 both directions. */
|
|
84
|
+
resolveSampleRate(_config) {
|
|
85
|
+
return XAI_PCM_SAMPLE_RATE;
|
|
217
86
|
}
|
|
218
|
-
/**
|
|
219
|
-
* @inheritdoc
|
|
220
|
-
*
|
|
221
|
-
* Sends `response.cancel` (only when a response is actually in flight) and flushes the
|
|
222
|
-
* locally-owned playout queue so already-generated speech stops coming out of the speaker
|
|
223
|
-
* immediately. Resets the local response state machine (active flag, narration kind,
|
|
224
|
-
* accumulated transcript) but PRESERVES any queued tool-result trigger: delegated work is
|
|
225
|
-
* never affected by a floor-control cancel, and the queued trigger still fires on the
|
|
226
|
-
* cancelled response's trailing `response.done`. No-op when idle or when the socket is gone.
|
|
227
|
-
*/
|
|
228
|
-
CancelActiveResponse() {
|
|
229
|
-
if (!this.socket) {
|
|
230
|
-
return;
|
|
231
|
-
}
|
|
232
|
-
if (!this.responseActive && !this.IsAudioPlaying) {
|
|
233
|
-
return; // nothing active — no-op by contract
|
|
234
|
-
}
|
|
235
|
-
if (this.responseActive) {
|
|
236
|
-
this.sendEvent({ type: 'response.cancel' });
|
|
237
|
-
this.responseActive = false;
|
|
238
|
-
this.pendingNarrationKind = false;
|
|
239
|
-
this.activeResponseKind = 'normal';
|
|
240
|
-
this.pendingAssistantText = '';
|
|
241
|
-
}
|
|
242
|
-
// The client OWNS the audio plane (websocket transport) — flush the local playout queue
|
|
243
|
-
// so speech stops immediately (obligation #3), no provider clear frame is needed.
|
|
244
|
-
this.playback?.Flush();
|
|
245
|
-
if (this.currentState === 'speaking') {
|
|
246
|
-
this.setState('listening');
|
|
247
|
-
}
|
|
248
|
-
}
|
|
249
|
-
/**
|
|
250
|
-
* Injects a system-role context item the model can draw on the next time it speaks, WITHOUT
|
|
251
|
-
* forcing a reply. Item creation is always safe mid-response, so it is sent immediately even
|
|
252
|
-
* while a reply is in flight.
|
|
253
|
-
*
|
|
254
|
-
* NOTE: role must be 'system' — the OpenAI-compatible realtime API rejects 'developer' items.
|
|
255
|
-
*/
|
|
256
|
-
SendContextNote(text) {
|
|
257
|
-
if (!this.socket) {
|
|
258
|
-
return;
|
|
259
|
-
}
|
|
260
|
-
this.sendEvent({
|
|
261
|
-
type: 'conversation.item.create',
|
|
262
|
-
item: {
|
|
263
|
-
type: 'message',
|
|
264
|
-
role: 'system',
|
|
265
|
-
content: [{ type: 'input_text', text }],
|
|
266
|
-
},
|
|
267
|
-
});
|
|
268
|
-
}
|
|
269
|
-
/**
|
|
270
|
-
* Triggers ONE short spoken update with the given instructions. Marks the upcoming response
|
|
271
|
-
* as `'narration'` (flag consumed by the next `response.created`) so its transcripts are
|
|
272
|
-
* emitted with `Kind: 'narration'` — ephemeral by contract. Sets {@link responseActive}
|
|
273
|
-
* eagerly so a tool result landing mid-narration queues instead of colliding.
|
|
274
|
-
*
|
|
275
|
-
* **Skips when busy** (base-contract collision rule — drivers MUST queue or skip): a
|
|
276
|
-
* `response.create` sent while a response is in flight would be rejected/garbled by the
|
|
277
|
-
* provider, and narration is disposable by contract, so the update is dropped with a debug
|
|
278
|
-
* log rather than queued to come out late and stale. Hosts SHOULD still gate on
|
|
279
|
-
* {@link IsBusy} / {@link IsAudioPlaying} for timing quality.
|
|
280
|
-
*/
|
|
281
|
-
RequestSpokenUpdate(instructions) {
|
|
282
|
-
if (!this.socket) {
|
|
283
|
-
return;
|
|
284
|
-
}
|
|
285
|
-
if (this.responseActive) {
|
|
286
|
-
console.debug('[xAIRealtimeClient] RequestSpokenUpdate skipped — a response is already in flight (narration is disposable).');
|
|
287
|
-
return;
|
|
288
|
-
}
|
|
289
|
-
this.responseActive = true;
|
|
290
|
-
this.pendingNarrationKind = true;
|
|
291
|
-
this.sendEvent({ type: 'response.create', response: { instructions } });
|
|
292
|
-
}
|
|
293
|
-
/**
|
|
294
|
-
* Sends the tool result back as a `function_call_output` conversation item, then triggers a
|
|
295
|
-
* reply — immediately if the model is idle, otherwise queued until the current response
|
|
296
|
-
* (e.g. a progress narration) finishes. Without the queueing the result's `response.create`
|
|
297
|
-
* would collide with an in-flight narration and be dropped, leaving the model silent when
|
|
298
|
-
* delegated work comes back.
|
|
299
|
-
*/
|
|
300
|
-
SendToolResult(callID, outputJson) {
|
|
301
|
-
if (!this.socket) {
|
|
302
|
-
return;
|
|
303
|
-
}
|
|
304
|
-
this.sendEvent({
|
|
305
|
-
type: 'conversation.item.create',
|
|
306
|
-
item: {
|
|
307
|
-
type: 'function_call_output',
|
|
308
|
-
call_id: callID,
|
|
309
|
-
output: outputJson,
|
|
310
|
-
},
|
|
311
|
-
});
|
|
312
|
-
this.requestResultResponse();
|
|
313
|
-
}
|
|
314
|
-
/**
|
|
315
|
-
* Mutes / unmutes by toggling the mic tracks' `enabled` flag: the capture pipeline stays up
|
|
316
|
-
* and streams SILENCE while muted (the provider's VAD sees a continuous stream and the
|
|
317
|
-
* un-mute is glitch-free — same policy as the other client drivers).
|
|
318
|
-
*/
|
|
319
|
-
SetMuted(muted) {
|
|
320
|
-
const tracks = this.micStream?.getAudioTracks() ?? [];
|
|
321
|
-
for (const track of tracks) {
|
|
322
|
-
track.enabled = !muted;
|
|
323
|
-
}
|
|
324
|
-
}
|
|
325
|
-
/** @inheritdoc */
|
|
326
|
-
get IsBusy() {
|
|
327
|
-
return this.responseActive;
|
|
328
|
-
}
|
|
329
|
-
/**
|
|
330
|
-
* @inheritdoc
|
|
331
|
-
*
|
|
332
|
-
* Computed directly from the playout engine's playhead clock — this client OWNS the output
|
|
333
|
-
* buffer, so "audibly playing" is precisely "scheduled audio extends beyond the audio
|
|
334
|
-
* context's current time".
|
|
335
|
-
*/
|
|
336
|
-
get IsAudioPlaying() {
|
|
337
|
-
return this.playback?.IsPlaying ?? false;
|
|
338
|
-
}
|
|
339
|
-
// ── Overridable creation seams (tests inject fakes — no network / audio) ──
|
|
340
87
|
/**
|
|
341
88
|
* Creation seam for the realtime websocket. Production wraps the platform-global `WebSocket`
|
|
342
89
|
* opened against the model-on-URL endpoint WITH the `xai-client-secret.<token>` subprotocol
|
|
343
90
|
* (browser auth — no handshake header is possible); unit tests override this to return an
|
|
344
|
-
* in-memory fake. Handlers are attached by
|
|
91
|
+
* in-memory fake. Handlers are attached by the shared Connect AFTER this returns, so the
|
|
345
92
|
* implementation must not require them at construction time.
|
|
346
93
|
*/
|
|
347
94
|
createSocket(url, subprotocol) {
|
|
@@ -364,301 +111,85 @@ let xAIRealtimeClient = class xAIRealtimeClient extends BaseRealtimeClient {
|
|
|
364
111
|
ws.onclose = () => seam.onclose?.();
|
|
365
112
|
return seam;
|
|
366
113
|
}
|
|
114
|
+
// ── Grok behavior overrides ────────────────────────────────────────────────
|
|
367
115
|
/**
|
|
368
|
-
*
|
|
369
|
-
*
|
|
370
|
-
*
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
/**
|
|
376
|
-
* Creation seam for the playout engine at the provider's fixed 24 kHz rate. Production
|
|
377
|
-
* returns the shared {@link RealtimePcmPlayback}.
|
|
378
|
-
*/
|
|
379
|
-
createPlayback(sampleRate) {
|
|
380
|
-
return new RealtimePcmPlayback(sampleRate);
|
|
381
|
-
}
|
|
382
|
-
// ── Connection internals ───────────────────────────────────────────────────
|
|
383
|
-
/**
|
|
384
|
-
* Sends the server-controlled session config (instructions + tools) as a `session.update` so
|
|
385
|
-
* the co-agent's identity and tool set apply. Skipped when the host supplied no config (e.g.
|
|
386
|
-
* it failed to parse the server payload — the host already logged that; sending an EMPTY
|
|
387
|
-
* `session.update` would be wrong).
|
|
388
|
-
*/
|
|
389
|
-
applySessionConfig() {
|
|
390
|
-
if (!this.sessionConfig || Object.keys(this.sessionConfig).length === 0) {
|
|
391
|
-
return;
|
|
392
|
-
}
|
|
393
|
-
this.sendEvent({ type: 'session.update', session: this.sessionConfig });
|
|
394
|
-
}
|
|
395
|
-
/** Streams one base64 PCM16 mic chunk as an `input_audio_buffer.append` frame. */
|
|
396
|
-
sendMicChunk(base64Pcm16) {
|
|
397
|
-
if (this.socket) {
|
|
398
|
-
this.sendEvent({ type: 'input_audio_buffer.append', audio: base64Pcm16 });
|
|
399
|
-
}
|
|
400
|
-
}
|
|
401
|
-
/** Surfaces a fatal socket error and marks the session unusable (obligation #6). */
|
|
402
|
-
handleSocketError(message) {
|
|
403
|
-
if (this.currentState === 'error' || this.currentState === 'closed') {
|
|
404
|
-
return;
|
|
405
|
-
}
|
|
406
|
-
this.emitError({ Message: `xAI Grok Voice realtime transport error: ${message}`, Fatal: true });
|
|
407
|
-
this.setState('error');
|
|
408
|
-
}
|
|
409
|
-
/**
|
|
410
|
-
* A socket close the CONSUMER didn't ask for is fatal: the provider hard-closes at token
|
|
411
|
-
* expiry and when it ends the session itself, so an unexpected close is how credential /
|
|
412
|
-
* session death reaches the host (obligation #6).
|
|
413
|
-
*/
|
|
414
|
-
handleSocketClose() {
|
|
415
|
-
if (this.closedByConsumer || this.currentState === 'error' || this.currentState === 'closed') {
|
|
416
|
-
return;
|
|
417
|
-
}
|
|
418
|
-
this.emitError({ Message: 'xAI Grok Voice realtime connection closed unexpectedly', Fatal: true });
|
|
419
|
-
this.setState('error');
|
|
420
|
-
}
|
|
421
|
-
// ── Inbound message translation ────────────────────────────────────────────
|
|
422
|
-
/** Parses one raw socket payload; non-JSON frames are ignored. */
|
|
423
|
-
handleSocketMessage(data) {
|
|
424
|
-
let event;
|
|
425
|
-
try {
|
|
426
|
-
event = JSON.parse(data);
|
|
427
|
-
}
|
|
428
|
-
catch {
|
|
429
|
-
console.debug('[xAIRealtimeClient] ◀ inbound NON-JSON frame:', String(data).slice(0, 200));
|
|
430
|
-
return; // non-JSON frame — ignore
|
|
431
|
-
}
|
|
432
|
-
if (event === null || typeof event !== 'object') {
|
|
433
|
-
return; // valid JSON but not an event object (e.g. "null", a number) — ignore
|
|
434
|
-
}
|
|
435
|
-
// DIAGNOSTIC: log every inbound event type so a "connected but silent" session is debuggable —
|
|
436
|
-
// surfaces error frames + any event whose name diverges from the OpenAI-compatible set (which
|
|
437
|
-
// the handleEvent switch would otherwise drop silently). Error frames also log their payload.
|
|
438
|
-
console.debug('[xAIRealtimeClient] ◀ inbound:', event.type, event.type === 'error' ? JSON.stringify(event).slice(0, 400) : '');
|
|
439
|
-
this.handleEvent(event);
|
|
440
|
-
}
|
|
441
|
-
/** Dispatches a typed xAI realtime server event to the appropriate behavior. */
|
|
442
|
-
handleEvent(event) {
|
|
443
|
-
switch (event.type) {
|
|
444
|
-
case 'response.output_audio_transcript.delta':
|
|
445
|
-
case 'response.audio_transcript.delta':
|
|
446
|
-
this.onAssistantDelta(event.delta);
|
|
447
|
-
break;
|
|
448
|
-
case 'response.output_audio_transcript.done':
|
|
449
|
-
case 'response.audio_transcript.done':
|
|
450
|
-
this.onAssistantDone(event.transcript);
|
|
451
|
-
break;
|
|
452
|
-
case 'response.audio.delta':
|
|
453
|
-
case 'response.output_audio.delta':
|
|
454
|
-
this.onAudioDelta(event.delta);
|
|
455
|
-
break;
|
|
456
|
-
case 'conversation.item.input_audio_transcription.completed':
|
|
457
|
-
this.onUserTranscript(event.transcript);
|
|
458
|
-
break;
|
|
459
|
-
case 'response.function_call_arguments.done':
|
|
460
|
-
this.onToolCallFrame(event);
|
|
461
|
-
break;
|
|
462
|
-
case 'input_audio_buffer.speech_started':
|
|
463
|
-
this.onSpeechStarted();
|
|
464
|
-
break;
|
|
465
|
-
case 'response.created':
|
|
466
|
-
this.responseActive = true;
|
|
467
|
-
// Stamp the kind of THIS response: 'narration' only when the flag was set by
|
|
468
|
-
// RequestSpokenUpdate immediately before its response.create (consumed here).
|
|
469
|
-
this.activeResponseKind = this.pendingNarrationKind ? 'narration' : 'normal';
|
|
470
|
-
this.pendingNarrationKind = false;
|
|
471
|
-
break;
|
|
472
|
-
case 'response.done':
|
|
473
|
-
// A turn finished — release the lock and speak any queued tool result so the
|
|
474
|
-
// model always voices the answer when delegated work comes back. The
|
|
475
|
-
// transcript-done frame for this turn has already arrived (it precedes
|
|
476
|
-
// response.done), so it's safe to reset the response kind here.
|
|
477
|
-
this.responseActive = false;
|
|
478
|
-
this.activeResponseKind = 'normal';
|
|
479
|
-
this.emitResponseUsage(event);
|
|
480
|
-
this.flushPendingResultResponse();
|
|
481
|
-
if (this.currentState === 'speaking') {
|
|
482
|
-
this.setState('listening');
|
|
483
|
-
}
|
|
484
|
-
break;
|
|
485
|
-
case 'error':
|
|
486
|
-
this.onErrorFrame(event);
|
|
487
|
-
break;
|
|
488
|
-
default:
|
|
489
|
-
// Unhandled event types are expected (the provider emits many); no-op.
|
|
490
|
-
break;
|
|
491
|
-
}
|
|
492
|
-
}
|
|
493
|
-
/** Appends an assistant transcript delta, reflects `'speaking'`, and emits the delta. */
|
|
494
|
-
onAssistantDelta(delta) {
|
|
495
|
-
if (this.currentState !== 'speaking') {
|
|
496
|
-
this.setState('speaking');
|
|
497
|
-
}
|
|
498
|
-
this.pendingAssistantText += delta;
|
|
499
|
-
this.emitTranscript({ Role: 'Assistant', Text: delta, IsFinal: false, Kind: this.activeResponseKind });
|
|
500
|
-
}
|
|
501
|
-
/**
|
|
502
|
-
* Finalizes the assistant turn: emits the final transcript tagged with the ACTIVE response
|
|
503
|
-
* kind (the transcript-done frame arrives BEFORE `response.done`, so
|
|
504
|
-
* {@link activeResponseKind} still reflects this turn), then returns to `'listening'`. Empty
|
|
505
|
-
* turns emit nothing.
|
|
506
|
-
*/
|
|
507
|
-
onAssistantDone(transcript) {
|
|
508
|
-
const finalText = transcript || this.pendingAssistantText;
|
|
509
|
-
this.pendingAssistantText = '';
|
|
510
|
-
if (finalText.trim().length > 0) {
|
|
511
|
-
this.emitTranscript({ Role: 'Assistant', Text: finalText, IsFinal: true, Kind: this.activeResponseKind });
|
|
512
|
-
}
|
|
513
|
-
if (this.currentState === 'speaking') {
|
|
514
|
-
this.setState('listening');
|
|
515
|
-
}
|
|
516
|
-
}
|
|
517
|
-
/**
|
|
518
|
-
* Decodes one base64 PCM16 chunk of the agent's spoken output into the playout queue and
|
|
519
|
-
* reflects `'speaking'`. The client OWNS the audio plane on this websocket transport (unlike
|
|
520
|
-
* the WebRTC OpenAI driver where the peer connection plays the remote track), so agent audio
|
|
521
|
-
* arrives as these deltas and is scheduled by the shared playout engine.
|
|
522
|
-
*/
|
|
523
|
-
onAudioDelta(deltaBase64) {
|
|
524
|
-
if (!deltaBase64) {
|
|
525
|
-
return;
|
|
526
|
-
}
|
|
527
|
-
if (this.currentState !== 'speaking') {
|
|
528
|
-
this.setState('speaking');
|
|
529
|
-
}
|
|
530
|
-
this.playback?.Enqueue(base64ToArrayBuffer(deltaBase64));
|
|
531
|
-
}
|
|
532
|
-
/**
|
|
533
|
-
* Emits the user's spoken-input transcription. Grok STREAMS this — repeated
|
|
534
|
-
* `input_audio_transcription.completed` events, each carrying the full text so far — unlike OpenAI's
|
|
535
|
-
* single final. So the FIRST emission of a turn appends a fresh caption and every later one is flagged
|
|
536
|
-
* ReplacesPrevious, collapsing the stream into ONE in-place-updating user bubble. The per-turn flag
|
|
537
|
-
* resets on the next `speech_started` ({@link onSpeechStarted}).
|
|
116
|
+
* @inheritdoc
|
|
117
|
+
*
|
|
118
|
+
* Grok STREAMS the input transcription — repeated `input_audio_transcription.completed`
|
|
119
|
+
* events, each carrying the full text so far — unlike OpenAI's single final. So the FIRST
|
|
120
|
+
* emission of a turn appends a fresh caption and every later one is flagged
|
|
121
|
+
* ReplacesPrevious, collapsing the stream into ONE in-place-updating user bubble. The
|
|
122
|
+
* per-turn flag resets on the next `speech_started` ({@link onSpeechStartedFrame}).
|
|
538
123
|
*/
|
|
539
|
-
|
|
124
|
+
onUserTranscriptFrame(transcript) {
|
|
540
125
|
if (transcript && transcript.trim().length > 0) {
|
|
126
|
+
// Two ways this caption REPLACES rather than appends: another already landed in this turn,
|
|
127
|
+
// OR it continues the previous utterance (the provider re-emits the whole thing with more
|
|
128
|
+
// words). The second case rescues a mid-sentence pause — the VAD fires speech_started and
|
|
129
|
+
// clears the flag even though the user never stopped, which would otherwise split one
|
|
130
|
+
// spoken thought into several growing bubbles. Mirrors the server session exactly so both
|
|
131
|
+
// topologies collapse the stream identically.
|
|
132
|
+
const replacesPrevious = this.userTurnTranscribed || IsTranscriptContinuation(this.lastUserTranscript, transcript);
|
|
541
133
|
this.emitTranscript({
|
|
542
134
|
Role: 'User', Text: transcript, IsFinal: true, Kind: 'normal',
|
|
543
|
-
ReplacesPrevious:
|
|
135
|
+
ReplacesPrevious: replacesPrevious,
|
|
544
136
|
});
|
|
545
137
|
this.userTurnTranscribed = true;
|
|
138
|
+
this.lastUserTranscript = transcript;
|
|
546
139
|
}
|
|
547
140
|
}
|
|
548
141
|
/**
|
|
549
|
-
*
|
|
550
|
-
*
|
|
551
|
-
*
|
|
552
|
-
*
|
|
553
|
-
*
|
|
142
|
+
* @inheritdoc
|
|
143
|
+
*
|
|
144
|
+
* On top of the shared websocket behavior (flush local playout on TRUE barge-in, gated
|
|
145
|
+
* interruption): a new user turn begins, so the streamed-transcription flag resets, and an
|
|
146
|
+
* interrupted response's busy flag is cleared eagerly (Grok cancels its own turn; clearing
|
|
147
|
+
* now lets a queued tool result fire without waiting on the trailing `response.done`).
|
|
554
148
|
*/
|
|
555
|
-
onToolCallFrame(call) {
|
|
556
|
-
if (this.currentState === 'speaking') {
|
|
557
|
-
this.currentState = 'connected';
|
|
558
|
-
}
|
|
559
|
-
this.responseActive = false;
|
|
560
|
-
this.emitToolCall({ CallID: call.call_id, ToolName: call.name, ArgumentsJson: call.arguments });
|
|
561
|
-
}
|
|
562
149
|
/**
|
|
563
|
-
*
|
|
564
|
-
*
|
|
565
|
-
*
|
|
566
|
-
*
|
|
567
|
-
*
|
|
568
|
-
* emits a terminal `response.done`.
|
|
150
|
+
* @inheritdoc
|
|
151
|
+
*
|
|
152
|
+
* The model has taken the floor, so the user's turn is definitively over — drop the continuation
|
|
153
|
+
* anchor. Without this, a later utterance that happened to open with the same words could be
|
|
154
|
+
* merged into the previous turn instead of appending its own.
|
|
569
155
|
*/
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
156
|
+
onResponseStarted() {
|
|
157
|
+
this.lastUserTranscript = '';
|
|
158
|
+
}
|
|
159
|
+
onSpeechStartedFrame() {
|
|
573
160
|
this.userTurnTranscribed = false;
|
|
574
|
-
if (
|
|
575
|
-
this.
|
|
576
|
-
|
|
161
|
+
if (this.responseActive || this.IsAudioPlaying) {
|
|
162
|
+
this.playback?.Flush();
|
|
163
|
+
this.responseActive = false;
|
|
164
|
+
this.activeResponseKind = 'normal';
|
|
165
|
+
// Floor to the user — drop the queued auto-trigger (see the brain's docstring).
|
|
166
|
+
this.pendingResultResponse = false;
|
|
167
|
+
this.emitInterruption();
|
|
577
168
|
}
|
|
578
|
-
this.playback?.Flush();
|
|
579
|
-
this.responseActive = false;
|
|
580
|
-
this.activeResponseKind = 'normal';
|
|
581
|
-
this.emitInterruption();
|
|
582
169
|
this.setState('listening');
|
|
583
170
|
}
|
|
584
|
-
/**
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
* contract's preferred shape). Frames without a usage payload emit nothing.
|
|
588
|
-
*/
|
|
589
|
-
emitResponseUsage(event) {
|
|
590
|
-
const usage = event.response?.usage;
|
|
591
|
-
if (!usage) {
|
|
592
|
-
return;
|
|
593
|
-
}
|
|
594
|
-
this.emitUsage({
|
|
595
|
-
InputTokens: typeof usage.input_tokens === 'number' ? usage.input_tokens : undefined,
|
|
596
|
-
OutputTokens: typeof usage.output_tokens === 'number' ? usage.output_tokens : undefined,
|
|
597
|
-
Raw: usage,
|
|
598
|
-
});
|
|
599
|
-
}
|
|
600
|
-
/** Surfaces a provider error frame (non-fatal; the session continues). */
|
|
601
|
-
onErrorFrame(event) {
|
|
602
|
-
this.emitError({
|
|
603
|
-
Message: event.error?.message ?? 'Unknown provider error',
|
|
604
|
-
Code: event.error?.code,
|
|
605
|
-
Fatal: false,
|
|
606
|
-
});
|
|
607
|
-
}
|
|
608
|
-
// ── Response state machine ─────────────────────────────────────────────────
|
|
609
|
-
/**
|
|
610
|
-
* Asks the model to speak (a tool result or typed-text reply) — immediately if it's idle,
|
|
611
|
-
* otherwise queued until the current response finishes. An immediate trigger also CONSUMES
|
|
612
|
-
* any queued trigger debt: every payload item is already in the conversation, so one
|
|
613
|
-
* `response.create` voices everything (e.g. typed text barging in over a narration that had
|
|
614
|
-
* tool results queued behind it).
|
|
615
|
-
*/
|
|
616
|
-
requestResultResponse() {
|
|
617
|
-
if (!this.socket) {
|
|
618
|
-
return;
|
|
619
|
-
}
|
|
620
|
-
if (this.responseActive) {
|
|
621
|
-
this.pendingResultResponse = true;
|
|
622
|
-
return;
|
|
623
|
-
}
|
|
624
|
-
this.pendingResultResponse = false;
|
|
625
|
-
this.responseActive = true;
|
|
626
|
-
this.sendEvent({ type: 'response.create' });
|
|
627
|
-
this.setState('speaking');
|
|
171
|
+
/** @inheritdoc — provider-branded transport-error message (kept stable for hosts/logs). */
|
|
172
|
+
formatTransportError(message) {
|
|
173
|
+
return `xAI Grok Voice realtime transport error: ${message}`;
|
|
628
174
|
}
|
|
629
|
-
/**
|
|
630
|
-
|
|
631
|
-
|
|
632
|
-
return;
|
|
633
|
-
}
|
|
634
|
-
this.pendingResultResponse = false;
|
|
635
|
-
this.responseActive = true;
|
|
636
|
-
this.sendEvent({ type: 'response.create' });
|
|
637
|
-
this.setState('speaking');
|
|
175
|
+
/** @inheritdoc — provider-branded unexpected-close message (kept stable for hosts/logs). */
|
|
176
|
+
get unexpectedCloseMessage() {
|
|
177
|
+
return 'xAI Grok Voice realtime connection closed unexpectedly';
|
|
638
178
|
}
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
this.pendingResultResponse = false;
|
|
644
|
-
this.pendingNarrationKind = false;
|
|
645
|
-
this.activeResponseKind = 'normal';
|
|
179
|
+
// ── Wire diagnostics (a silent Grok session is otherwise undebuggable) ────
|
|
180
|
+
/** @inheritdoc — log every inbound event type; error frames include their payload. */
|
|
181
|
+
logInboundEvent(event) {
|
|
182
|
+
console.debug('[xAIRealtimeClient] ◀ inbound:', event.type, event.type === 'error' ? JSON.stringify(event).slice(0, 400) : '');
|
|
646
183
|
}
|
|
647
|
-
|
|
648
|
-
|
|
649
|
-
|
|
650
|
-
this.currentState = state;
|
|
651
|
-
this.emitStateChange(state);
|
|
184
|
+
/** @inheritdoc */
|
|
185
|
+
onNonJsonFrame(raw) {
|
|
186
|
+
console.debug('[xAIRealtimeClient] ◀ inbound NON-JSON frame:', String(raw).slice(0, 200));
|
|
652
187
|
}
|
|
653
|
-
/**
|
|
654
|
-
|
|
655
|
-
// DIAGNOSTIC: log outbound control frames (skip the high-frequency mic audio) so we can see
|
|
656
|
-
// session.update / conversation.item.create / response.create actually went out — the other
|
|
657
|
-
// half of diagnosing a silent session.
|
|
188
|
+
/** @inheritdoc — log outbound control frames (skip the high-frequency mic audio). */
|
|
189
|
+
logOutboundEvent(event) {
|
|
658
190
|
if (event.type !== 'input_audio_buffer.append') {
|
|
659
191
|
console.debug('[xAIRealtimeClient] ▶ outbound:', event.type);
|
|
660
192
|
}
|
|
661
|
-
this.socket?.send(JSON.stringify(event));
|
|
662
193
|
}
|
|
663
194
|
};
|
|
664
195
|
xAIRealtimeClient = __decorate([
|