@memberjunction/ai-realtime-client 0.0.1 → 5.42.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +148 -28
- package/dist/audio/audioMeter.d.ts +101 -0
- package/dist/audio/audioMeter.d.ts.map +1 -0
- package/dist/audio/audioMeter.js +193 -0
- package/dist/audio/audioMeter.js.map +1 -0
- package/dist/audio/micCapture.d.ts +26 -0
- package/dist/audio/micCapture.d.ts.map +1 -0
- package/dist/audio/micCapture.js +69 -0
- package/dist/audio/micCapture.js.map +1 -0
- package/dist/audio/pcmPlayback.d.ts +73 -0
- package/dist/audio/pcmPlayback.d.ts.map +1 -0
- package/dist/audio/pcmPlayback.js +78 -0
- package/dist/audio/pcmPlayback.js.map +1 -0
- package/dist/audio/pcmUtils.d.ts +17 -0
- package/dist/audio/pcmUtils.d.ts.map +1 -0
- package/dist/audio/pcmUtils.js +46 -0
- package/dist/audio/pcmUtils.js.map +1 -0
- package/dist/drivers/assemblyAIRealtimeClient.d.ts +384 -0
- package/dist/drivers/assemblyAIRealtimeClient.d.ts.map +1 -0
- package/dist/drivers/assemblyAIRealtimeClient.js +732 -0
- package/dist/drivers/assemblyAIRealtimeClient.js.map +1 -0
- package/dist/drivers/elevenLabsRealtimeClient.d.ts +362 -0
- package/dist/drivers/elevenLabsRealtimeClient.d.ts.map +1 -0
- package/dist/drivers/elevenLabsRealtimeClient.js +686 -0
- package/dist/drivers/elevenLabsRealtimeClient.js.map +1 -0
- package/dist/drivers/geminiRealtimeClient.d.ts +406 -0
- package/dist/drivers/geminiRealtimeClient.d.ts.map +1 -0
- package/dist/drivers/geminiRealtimeClient.js +675 -0
- package/dist/drivers/geminiRealtimeClient.js.map +1 -0
- package/dist/drivers/openAIRealtimeClient.d.ts +381 -0
- package/dist/drivers/openAIRealtimeClient.d.ts.map +1 -0
- package/dist/drivers/openAIRealtimeClient.js +602 -0
- package/dist/drivers/openAIRealtimeClient.js.map +1 -0
- package/dist/drivers/xaiRealtimeClient.d.ts +430 -0
- package/dist/drivers/xaiRealtimeClient.d.ts.map +1 -0
- package/dist/drivers/xaiRealtimeClient.js +676 -0
- package/dist/drivers/xaiRealtimeClient.js.map +1 -0
- package/dist/generic/baseRealtimeClient.d.ts +401 -0
- package/dist/generic/baseRealtimeClient.d.ts.map +1 -0
- package/dist/generic/baseRealtimeClient.js +226 -0
- package/dist/generic/baseRealtimeClient.js.map +1 -0
- package/dist/index.d.ts +11 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11 -0
- package/dist/index.js.map +1 -0
- package/package.json +28 -7
|
@@ -0,0 +1,676 @@
|
|
|
1
|
+
var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
|
|
2
|
+
var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
|
|
3
|
+
if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
|
|
4
|
+
else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
|
|
5
|
+
return c > 3 && r && Object.defineProperty(target, key, r), r;
|
|
6
|
+
};
|
|
7
|
+
import { RegisterClass } from '@memberjunction/global';
|
|
8
|
+
import { BaseRealtimeClient } from '../generic/baseRealtimeClient.js';
|
|
9
|
+
import { base64ToArrayBuffer } from '../audio/pcmUtils.js';
|
|
10
|
+
import { RealtimePcmPlayback } from '../audio/pcmPlayback.js';
|
|
11
|
+
import { RealtimeAudioMeter } from '../audio/audioMeter.js';
|
|
12
|
+
import { createPcmMicCapture } from '../audio/micCapture.js';
|
|
13
|
+
// ── Audio + endpoint constants (xAI Grok Voice wire format) ────────────────────
|
|
14
|
+
/**
|
|
15
|
+
* The Grok Voice realtime websocket endpoint. The model is appended as a `?model=` query
|
|
16
|
+
* parameter, derived from `config.Model` (e.g. `grok-voice-latest`) — unlike OpenAI's GA
|
|
17
|
+
* browser flow (where the ephemeral secret encodes the model), xAI takes the model on the URL.
|
|
18
|
+
*/
|
|
19
|
+
export const XAI_REALTIME_WS_URL = 'wss://api.x.ai/v1/realtime';
|
|
20
|
+
/**
|
|
21
|
+
* Browser auth rides as a websocket SUBPROTOCOL with this prefix: the ephemeral client secret
|
|
22
|
+
* is passed as `"xai-client-secret." + token`. Browsers cannot set request headers on a
|
|
23
|
+
* websocket handshake, so an `Authorization` header is impossible — the subprotocol channel is
|
|
24
|
+
* how the server-minted one-time credential reaches xAI (no API key ever touches the browser).
|
|
25
|
+
*/
|
|
26
|
+
export const XAI_CLIENT_SECRET_SUBPROTOCOL_PREFIX = 'xai-client-secret.';
|
|
27
|
+
/**
|
|
28
|
+
* The Grok Voice wire audio format is FIXED: 16-bit signed little-endian PCM, mono, 24 kHz,
|
|
29
|
+
* base64-encoded, in BOTH directions (`input_audio_buffer.append` up, `response.audio.delta`
|
|
30
|
+
* down) — there is no per-session format negotiation on this provider.
|
|
31
|
+
*/
|
|
32
|
+
export const XAI_PCM_SAMPLE_RATE = 24000;
|
|
33
|
+
// ── The driver ─────────────────────────────────────────────────────────────────
|
|
34
|
+
/**
|
|
35
|
+
* xAI Grok Voice implementation of {@link BaseRealtimeClient}: a **browser-direct** websocket
|
|
36
|
+
* connection to xAI's Grok Voice realtime API, authenticated with the server-minted ONE-TIME
|
|
37
|
+
* ephemeral client secret passed as a websocket SUBPROTOCOL (no API key ever reaches the
|
|
38
|
+
* browser, and browsers cannot set a handshake `Authorization` header).
|
|
39
|
+
*
|
|
40
|
+
* Registered with the ClassFactory under the key `'xai'` — the `Provider` string the server's
|
|
41
|
+
* matching Grok Voice driver stamps on its `ClientRealtimeSessionConfig` — so hosts resolve it
|
|
42
|
+
* without referencing this class directly.
|
|
43
|
+
*
|
|
44
|
+
* It is the structural CROSS of the two sibling drivers:
|
|
45
|
+
* - Like the AssemblyAI client driver, the transport is a WEBSOCKET and the audio plane is
|
|
46
|
+
* CLIENT-OWNED: mic PCM16 streams out as base64 frames via the shared {@link createPcmMicCapture}
|
|
47
|
+
* worklet, and the agent's base64 audio chunks feed the shared {@link RealtimePcmPlayback}
|
|
48
|
+
* engine ({@link IsAudioPlaying} comes from its playout clock). There is NO SDP handshake, NO
|
|
49
|
+
* peer connection, NO remote-audio sink element.
|
|
50
|
+
* - Like the OpenAI client driver, the EVENT PROTOCOL is OpenAI-Realtime-API compatible: the
|
|
51
|
+
* same client events (session.update, conversation.item.create, response.create,
|
|
52
|
+
* response.cancel, input_audio_buffer.append) and the same server events
|
|
53
|
+
* (response.output_audio_transcript.delta and .done, response.audio.delta,
|
|
54
|
+
* conversation.item.input_audio_transcription.completed,
|
|
55
|
+
* response.function_call_arguments.done, response.created, response.done,
|
|
56
|
+
* input_audio_buffer.speech_started, error). The response state machine, narration-kind
|
|
57
|
+
* tagging, and tool-result queueing are identical to the OpenAI driver.
|
|
58
|
+
*
|
|
59
|
+
* Connect handshake: open the model-on-URL endpoint with the subprotocol auth → on socket open,
|
|
60
|
+
* apply the server-authored `config.SessionConfig` via a `session.update` frame → build the
|
|
61
|
+
* audio plane at the provider's FIXED 24 kHz PCM16 format → report `'listening'`. The
|
|
62
|
+
* `'listening'` gate is the socket-open + session.update applied point (obligation #7): unlike
|
|
63
|
+
* AssemblyAI there is no separate `session.ready` confirmation in this protocol, so applying the
|
|
64
|
+
* config on open is the readiness boundary.
|
|
65
|
+
*/
|
|
66
|
+
let xAIRealtimeClient = class xAIRealtimeClient extends BaseRealtimeClient {
|
|
67
|
+
constructor() {
|
|
68
|
+
super(...arguments);
|
|
69
|
+
// ── Transport / audio resources ────────────────────────────────────────────
|
|
70
|
+
this.socket = null;
|
|
71
|
+
this.micStream = null;
|
|
72
|
+
this.micCapture = null;
|
|
73
|
+
this.playback = null;
|
|
74
|
+
/**
|
|
75
|
+
* The server-built session config applied verbatim via `session.update` once the socket
|
|
76
|
+
* opens. Protected so test subclasses can seed it without a full Connect.
|
|
77
|
+
*/
|
|
78
|
+
this.sessionConfig = null;
|
|
79
|
+
// ── Response state machine (mirrors the OpenAI client driver) ──────────────
|
|
80
|
+
/**
|
|
81
|
+
* Whether the CURRENT user turn has already emitted a transcription. Grok streams input
|
|
82
|
+
* transcription as repeated `.completed` events (each the full growing text), so the first emission
|
|
83
|
+
* of a turn appends a caption and the rest are flagged ReplacesPrevious — one in-place bubble, not a
|
|
84
|
+
* stack of growing duplicates. Reset on each `input_audio_buffer.speech_started` (new turn).
|
|
85
|
+
*/
|
|
86
|
+
this.userTurnTranscribed = false;
|
|
87
|
+
/** Accumulates the in-flight assistant transcript across delta frames. */
|
|
88
|
+
this.pendingAssistantText = '';
|
|
89
|
+
/** True while the model has a response in flight; gates narration + queues the tool result. */
|
|
90
|
+
this.responseActive = false;
|
|
91
|
+
/** Set when a tool result is ready while a response is active; sent on the next response.done. */
|
|
92
|
+
this.pendingResultResponse = false;
|
|
93
|
+
/**
|
|
94
|
+
* Set by {@link RequestSpokenUpdate} just before it sends its `response.create`, and
|
|
95
|
+
* CONSUMED by the very next `response.created` frame, which stamps
|
|
96
|
+
* {@link activeResponseKind} for that turn.
|
|
97
|
+
*/
|
|
98
|
+
this.pendingNarrationKind = false;
|
|
99
|
+
/**
|
|
100
|
+
* The kind of the response currently in flight. Event ordering: `response.created` →
|
|
101
|
+
* transcript deltas → `*_audio_transcript.done` → `response.done`. The transcript-done
|
|
102
|
+
* frame therefore arrives while the kind is still set, letting {@link onAssistantDone}
|
|
103
|
+
* classify the turn; `response.done` then resets the kind to `'normal'`.
|
|
104
|
+
*/
|
|
105
|
+
this.activeResponseKind = 'normal';
|
|
106
|
+
/**
|
|
107
|
+
* The client's own view of the session state — mirrors what {@link emitStateChange} last
|
|
108
|
+
* reported, EXCEPT after a tool call is emitted: the host typically shows its own busy
|
|
109
|
+
* indicator then, so the client silently leaves `'speaking'` (no emission) to preserve the
|
|
110
|
+
* host's indicator until the result reply starts (see {@link handleEvent}).
|
|
111
|
+
*/
|
|
112
|
+
this.currentState = 'closed';
|
|
113
|
+
/** True once Disconnect ran — an expected socket close must not surface as fatal. */
|
|
114
|
+
this.closedByConsumer = false;
|
|
115
|
+
}
|
|
116
|
+
// ── BaseRealtimeClient: connection lifecycle ───────────────────────────────
|
|
117
|
+
/**
|
|
118
|
+
* Opens the client-direct Grok Voice websocket: the model-on-URL endpoint with the
|
|
119
|
+
* subprotocol auth, the server-authored `config.SessionConfig` applied via `session.update`
|
|
120
|
+
* once the socket opens, then the audio plane at the provider's fixed 24 kHz format. Reports
|
|
121
|
+
* `'listening'` only after all of that (obligation #7).
|
|
122
|
+
*/
|
|
123
|
+
async Connect(config, micStream) {
|
|
124
|
+
this.sessionConfig = config.SessionConfig;
|
|
125
|
+
this.micStream = micStream;
|
|
126
|
+
this.closedByConsumer = false;
|
|
127
|
+
this.setState('connecting');
|
|
128
|
+
let openSocket = null;
|
|
129
|
+
let failOpen = null;
|
|
130
|
+
const opened = new Promise((resolve, reject) => {
|
|
131
|
+
openSocket = resolve;
|
|
132
|
+
failOpen = reject;
|
|
133
|
+
});
|
|
134
|
+
const url = `${XAI_REALTIME_WS_URL}?model=${encodeURIComponent(config.Model)}`;
|
|
135
|
+
const subprotocol = `${XAI_CLIENT_SECRET_SUBPROTOCOL_PREFIX}${config.EphemeralToken}`;
|
|
136
|
+
const socket = this.createSocket(url, subprotocol);
|
|
137
|
+
this.socket = socket;
|
|
138
|
+
socket.onopen = () => openSocket?.();
|
|
139
|
+
socket.onmessage = (data) => this.handleSocketMessage(data);
|
|
140
|
+
socket.onerror = (message) => {
|
|
141
|
+
failOpen?.(new Error(message));
|
|
142
|
+
this.handleSocketError(message);
|
|
143
|
+
};
|
|
144
|
+
socket.onclose = () => {
|
|
145
|
+
failOpen?.(new Error('xAI Grok Voice socket closed during connect'));
|
|
146
|
+
this.handleSocketClose();
|
|
147
|
+
};
|
|
148
|
+
await opened;
|
|
149
|
+
this.setState('connected');
|
|
150
|
+
// The server-authored session config (the SessionConfig pact) is applied as the FIRST
|
|
151
|
+
// frame — prompt and tool authority stay server-side (obligation #8). This protocol has
|
|
152
|
+
// no separate readiness ack, so applying the config on open IS the readiness boundary.
|
|
153
|
+
this.applySessionConfig();
|
|
154
|
+
this.playback = this.createPlayback(XAI_PCM_SAMPLE_RATE);
|
|
155
|
+
this.micCapture = await this.createMicCapture(micStream, XAI_PCM_SAMPLE_RATE, (base64Pcm16) => this.sendMicChunk(base64Pcm16));
|
|
156
|
+
// Audio-activity capability (base obligation #9): agent side taps the playout engine's
|
|
157
|
+
// master gain; user side meters the mic stream. Null-safe — test fakes / no-WebAudio
|
|
158
|
+
// environments simply leave the session un-metered.
|
|
159
|
+
this.attachOutputAudioMeter(this.playback?.CreateMeter?.() ?? null);
|
|
160
|
+
this.attachInputAudioMeter(RealtimeAudioMeter.ForMicStream(micStream));
|
|
161
|
+
this.setState('listening');
|
|
162
|
+
}
|
|
163
|
+
/**
|
|
164
|
+
* Tears down the socket, mic capture, mic tracks, and playout engine, resets the response
|
|
165
|
+
* state machine, and emits a final `'closed'` (unless already `'error'`). Safe to call more
|
|
166
|
+
* than once.
|
|
167
|
+
*/
|
|
168
|
+
async Disconnect() {
|
|
169
|
+
this.closedByConsumer = true;
|
|
170
|
+
this.closeAudioMeters();
|
|
171
|
+
this.micStream?.getTracks().forEach((track) => track.stop());
|
|
172
|
+
this.micStream = null;
|
|
173
|
+
this.micCapture?.Stop();
|
|
174
|
+
this.micCapture = null;
|
|
175
|
+
this.playback?.Close();
|
|
176
|
+
this.playback = null;
|
|
177
|
+
if (this.socket) {
|
|
178
|
+
try {
|
|
179
|
+
this.socket.close();
|
|
180
|
+
}
|
|
181
|
+
catch {
|
|
182
|
+
/* already closing */
|
|
183
|
+
}
|
|
184
|
+
this.socket = null;
|
|
185
|
+
}
|
|
186
|
+
this.sessionConfig = null;
|
|
187
|
+
this.resetResponseState();
|
|
188
|
+
if (this.currentState !== 'error') {
|
|
189
|
+
this.setState('closed');
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
// ── BaseRealtimeClient: outbound actions ──────────────────────────────────
|
|
193
|
+
/**
|
|
194
|
+
* Injects typed text as a user-role `message` conversation item, then triggers a reply
|
|
195
|
+
* through the SAME collision-safe path tool results use ({@link requestResultResponse}).
|
|
196
|
+
* No-op when the socket isn't open.
|
|
197
|
+
*
|
|
198
|
+
* **SendText implies barge-in** (base-contract rule): an active spoken response is cancelled
|
|
199
|
+
* via {@link CancelActiveResponse} before the text is injected, so the typed turn takes the
|
|
200
|
+
* floor immediately instead of waiting behind stale speech. When nothing is active the cancel
|
|
201
|
+
* is a no-op and the reply triggers immediately.
|
|
202
|
+
*/
|
|
203
|
+
SendText(text) {
|
|
204
|
+
if (!this.socket) {
|
|
205
|
+
return;
|
|
206
|
+
}
|
|
207
|
+
this.CancelActiveResponse();
|
|
208
|
+
this.sendEvent({
|
|
209
|
+
type: 'conversation.item.create',
|
|
210
|
+
item: {
|
|
211
|
+
type: 'message',
|
|
212
|
+
role: 'user',
|
|
213
|
+
content: [{ type: 'input_text', text }],
|
|
214
|
+
},
|
|
215
|
+
});
|
|
216
|
+
this.requestResultResponse();
|
|
217
|
+
}
|
|
218
|
+
/**
|
|
219
|
+
* @inheritdoc
|
|
220
|
+
*
|
|
221
|
+
* Sends `response.cancel` (only when a response is actually in flight) and flushes the
|
|
222
|
+
* locally-owned playout queue so already-generated speech stops coming out of the speaker
|
|
223
|
+
* immediately. Resets the local response state machine (active flag, narration kind,
|
|
224
|
+
* accumulated transcript) but PRESERVES any queued tool-result trigger: delegated work is
|
|
225
|
+
* never affected by a floor-control cancel, and the queued trigger still fires on the
|
|
226
|
+
* cancelled response's trailing `response.done`. No-op when idle or when the socket is gone.
|
|
227
|
+
*/
|
|
228
|
+
CancelActiveResponse() {
|
|
229
|
+
if (!this.socket) {
|
|
230
|
+
return;
|
|
231
|
+
}
|
|
232
|
+
if (!this.responseActive && !this.IsAudioPlaying) {
|
|
233
|
+
return; // nothing active — no-op by contract
|
|
234
|
+
}
|
|
235
|
+
if (this.responseActive) {
|
|
236
|
+
this.sendEvent({ type: 'response.cancel' });
|
|
237
|
+
this.responseActive = false;
|
|
238
|
+
this.pendingNarrationKind = false;
|
|
239
|
+
this.activeResponseKind = 'normal';
|
|
240
|
+
this.pendingAssistantText = '';
|
|
241
|
+
}
|
|
242
|
+
// The client OWNS the audio plane (websocket transport) — flush the local playout queue
|
|
243
|
+
// so speech stops immediately (obligation #3), no provider clear frame is needed.
|
|
244
|
+
this.playback?.Flush();
|
|
245
|
+
if (this.currentState === 'speaking') {
|
|
246
|
+
this.setState('listening');
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
/**
|
|
250
|
+
* Injects a system-role context item the model can draw on the next time it speaks, WITHOUT
|
|
251
|
+
* forcing a reply. Item creation is always safe mid-response, so it is sent immediately even
|
|
252
|
+
* while a reply is in flight.
|
|
253
|
+
*
|
|
254
|
+
* NOTE: role must be 'system' — the OpenAI-compatible realtime API rejects 'developer' items.
|
|
255
|
+
*/
|
|
256
|
+
SendContextNote(text) {
|
|
257
|
+
if (!this.socket) {
|
|
258
|
+
return;
|
|
259
|
+
}
|
|
260
|
+
this.sendEvent({
|
|
261
|
+
type: 'conversation.item.create',
|
|
262
|
+
item: {
|
|
263
|
+
type: 'message',
|
|
264
|
+
role: 'system',
|
|
265
|
+
content: [{ type: 'input_text', text }],
|
|
266
|
+
},
|
|
267
|
+
});
|
|
268
|
+
}
|
|
269
|
+
/**
|
|
270
|
+
* Triggers ONE short spoken update with the given instructions. Marks the upcoming response
|
|
271
|
+
* as `'narration'` (flag consumed by the next `response.created`) so its transcripts are
|
|
272
|
+
* emitted with `Kind: 'narration'` — ephemeral by contract. Sets {@link responseActive}
|
|
273
|
+
* eagerly so a tool result landing mid-narration queues instead of colliding.
|
|
274
|
+
*
|
|
275
|
+
* **Skips when busy** (base-contract collision rule — drivers MUST queue or skip): a
|
|
276
|
+
* `response.create` sent while a response is in flight would be rejected/garbled by the
|
|
277
|
+
* provider, and narration is disposable by contract, so the update is dropped with a debug
|
|
278
|
+
* log rather than queued to come out late and stale. Hosts SHOULD still gate on
|
|
279
|
+
* {@link IsBusy} / {@link IsAudioPlaying} for timing quality.
|
|
280
|
+
*/
|
|
281
|
+
RequestSpokenUpdate(instructions) {
|
|
282
|
+
if (!this.socket) {
|
|
283
|
+
return;
|
|
284
|
+
}
|
|
285
|
+
if (this.responseActive) {
|
|
286
|
+
console.debug('[xAIRealtimeClient] RequestSpokenUpdate skipped — a response is already in flight (narration is disposable).');
|
|
287
|
+
return;
|
|
288
|
+
}
|
|
289
|
+
this.responseActive = true;
|
|
290
|
+
this.pendingNarrationKind = true;
|
|
291
|
+
this.sendEvent({ type: 'response.create', response: { instructions } });
|
|
292
|
+
}
|
|
293
|
+
/**
|
|
294
|
+
* Sends the tool result back as a `function_call_output` conversation item, then triggers a
|
|
295
|
+
* reply — immediately if the model is idle, otherwise queued until the current response
|
|
296
|
+
* (e.g. a progress narration) finishes. Without the queueing the result's `response.create`
|
|
297
|
+
* would collide with an in-flight narration and be dropped, leaving the model silent when
|
|
298
|
+
* delegated work comes back.
|
|
299
|
+
*/
|
|
300
|
+
SendToolResult(callID, outputJson) {
|
|
301
|
+
if (!this.socket) {
|
|
302
|
+
return;
|
|
303
|
+
}
|
|
304
|
+
this.sendEvent({
|
|
305
|
+
type: 'conversation.item.create',
|
|
306
|
+
item: {
|
|
307
|
+
type: 'function_call_output',
|
|
308
|
+
call_id: callID,
|
|
309
|
+
output: outputJson,
|
|
310
|
+
},
|
|
311
|
+
});
|
|
312
|
+
this.requestResultResponse();
|
|
313
|
+
}
|
|
314
|
+
/**
|
|
315
|
+
* Mutes / unmutes by toggling the mic tracks' `enabled` flag: the capture pipeline stays up
|
|
316
|
+
* and streams SILENCE while muted (the provider's VAD sees a continuous stream and the
|
|
317
|
+
* un-mute is glitch-free — same policy as the other client drivers).
|
|
318
|
+
*/
|
|
319
|
+
SetMuted(muted) {
|
|
320
|
+
const tracks = this.micStream?.getAudioTracks() ?? [];
|
|
321
|
+
for (const track of tracks) {
|
|
322
|
+
track.enabled = !muted;
|
|
323
|
+
}
|
|
324
|
+
}
|
|
325
|
+
/** @inheritdoc */
|
|
326
|
+
get IsBusy() {
|
|
327
|
+
return this.responseActive;
|
|
328
|
+
}
|
|
329
|
+
/**
|
|
330
|
+
* @inheritdoc
|
|
331
|
+
*
|
|
332
|
+
* Computed directly from the playout engine's playhead clock — this client OWNS the output
|
|
333
|
+
* buffer, so "audibly playing" is precisely "scheduled audio extends beyond the audio
|
|
334
|
+
* context's current time".
|
|
335
|
+
*/
|
|
336
|
+
get IsAudioPlaying() {
|
|
337
|
+
return this.playback?.IsPlaying ?? false;
|
|
338
|
+
}
|
|
339
|
+
// ── Overridable creation seams (tests inject fakes — no network / audio) ──
|
|
340
|
+
/**
|
|
341
|
+
* Creation seam for the realtime websocket. Production wraps the platform-global `WebSocket`
|
|
342
|
+
* opened against the model-on-URL endpoint WITH the `xai-client-secret.<token>` subprotocol
|
|
343
|
+
* (browser auth — no handshake header is possible); unit tests override this to return an
|
|
344
|
+
* in-memory fake. Handlers are attached by {@link Connect} AFTER this returns, so the
|
|
345
|
+
* implementation must not require them at construction time.
|
|
346
|
+
*/
|
|
347
|
+
createSocket(url, subprotocol) {
|
|
348
|
+
const WS = globalThis.WebSocket;
|
|
349
|
+
if (!WS) {
|
|
350
|
+
throw new Error('xAIRealtimeClient requires a global WebSocket (browser or Node 22+).');
|
|
351
|
+
}
|
|
352
|
+
const ws = new WS(url, subprotocol);
|
|
353
|
+
const seam = {
|
|
354
|
+
onopen: null,
|
|
355
|
+
onmessage: null,
|
|
356
|
+
onerror: null,
|
|
357
|
+
onclose: null,
|
|
358
|
+
send: (data) => ws.send(data),
|
|
359
|
+
close: () => ws.close(),
|
|
360
|
+
};
|
|
361
|
+
ws.onopen = () => seam.onopen?.();
|
|
362
|
+
ws.onmessage = (event) => seam.onmessage?.(String(event.data));
|
|
363
|
+
ws.onerror = () => seam.onerror?.('xAI Grok Voice websocket error');
|
|
364
|
+
ws.onclose = () => seam.onclose?.();
|
|
365
|
+
return seam;
|
|
366
|
+
}
|
|
367
|
+
/**
|
|
368
|
+
* Creation seam for the mic-capture pipeline at the provider's fixed 24 kHz rate. Production
|
|
369
|
+
* delegates to the shared {@link createPcmMicCapture}; unit tests override this with a no-op
|
|
370
|
+
* fake (and may capture `onPcmChunk` to simulate mic frames).
|
|
371
|
+
*/
|
|
372
|
+
async createMicCapture(micStream, sampleRate, onPcmChunk) {
|
|
373
|
+
return createPcmMicCapture(micStream, sampleRate, onPcmChunk);
|
|
374
|
+
}
|
|
375
|
+
/**
|
|
376
|
+
* Creation seam for the playout engine at the provider's fixed 24 kHz rate. Production
|
|
377
|
+
* returns the shared {@link RealtimePcmPlayback}.
|
|
378
|
+
*/
|
|
379
|
+
createPlayback(sampleRate) {
|
|
380
|
+
return new RealtimePcmPlayback(sampleRate);
|
|
381
|
+
}
|
|
382
|
+
// ── Connection internals ───────────────────────────────────────────────────
|
|
383
|
+
/**
|
|
384
|
+
* Sends the server-controlled session config (instructions + tools) as a `session.update` so
|
|
385
|
+
* the co-agent's identity and tool set apply. Skipped when the host supplied no config (e.g.
|
|
386
|
+
* it failed to parse the server payload — the host already logged that; sending an EMPTY
|
|
387
|
+
* `session.update` would be wrong).
|
|
388
|
+
*/
|
|
389
|
+
applySessionConfig() {
|
|
390
|
+
if (!this.sessionConfig || Object.keys(this.sessionConfig).length === 0) {
|
|
391
|
+
return;
|
|
392
|
+
}
|
|
393
|
+
this.sendEvent({ type: 'session.update', session: this.sessionConfig });
|
|
394
|
+
}
|
|
395
|
+
/** Streams one base64 PCM16 mic chunk as an `input_audio_buffer.append` frame. */
|
|
396
|
+
sendMicChunk(base64Pcm16) {
|
|
397
|
+
if (this.socket) {
|
|
398
|
+
this.sendEvent({ type: 'input_audio_buffer.append', audio: base64Pcm16 });
|
|
399
|
+
}
|
|
400
|
+
}
|
|
401
|
+
/** Surfaces a fatal socket error and marks the session unusable (obligation #6). */
|
|
402
|
+
handleSocketError(message) {
|
|
403
|
+
if (this.currentState === 'error' || this.currentState === 'closed') {
|
|
404
|
+
return;
|
|
405
|
+
}
|
|
406
|
+
this.emitError({ Message: `xAI Grok Voice realtime transport error: ${message}`, Fatal: true });
|
|
407
|
+
this.setState('error');
|
|
408
|
+
}
|
|
409
|
+
/**
|
|
410
|
+
* A socket close the CONSUMER didn't ask for is fatal: the provider hard-closes at token
|
|
411
|
+
* expiry and when it ends the session itself, so an unexpected close is how credential /
|
|
412
|
+
* session death reaches the host (obligation #6).
|
|
413
|
+
*/
|
|
414
|
+
handleSocketClose() {
|
|
415
|
+
if (this.closedByConsumer || this.currentState === 'error' || this.currentState === 'closed') {
|
|
416
|
+
return;
|
|
417
|
+
}
|
|
418
|
+
this.emitError({ Message: 'xAI Grok Voice realtime connection closed unexpectedly', Fatal: true });
|
|
419
|
+
this.setState('error');
|
|
420
|
+
}
|
|
421
|
+
// ── Inbound message translation ────────────────────────────────────────────
|
|
422
|
+
/** Parses one raw socket payload; non-JSON frames are ignored. */
|
|
423
|
+
handleSocketMessage(data) {
|
|
424
|
+
let event;
|
|
425
|
+
try {
|
|
426
|
+
event = JSON.parse(data);
|
|
427
|
+
}
|
|
428
|
+
catch {
|
|
429
|
+
console.debug('[xAIRealtimeClient] ◀ inbound NON-JSON frame:', String(data).slice(0, 200));
|
|
430
|
+
return; // non-JSON frame — ignore
|
|
431
|
+
}
|
|
432
|
+
if (event === null || typeof event !== 'object') {
|
|
433
|
+
return; // valid JSON but not an event object (e.g. "null", a number) — ignore
|
|
434
|
+
}
|
|
435
|
+
// DIAGNOSTIC: log every inbound event type so a "connected but silent" session is debuggable —
|
|
436
|
+
// surfaces error frames + any event whose name diverges from the OpenAI-compatible set (which
|
|
437
|
+
// the handleEvent switch would otherwise drop silently). Error frames also log their payload.
|
|
438
|
+
console.debug('[xAIRealtimeClient] ◀ inbound:', event.type, event.type === 'error' ? JSON.stringify(event).slice(0, 400) : '');
|
|
439
|
+
this.handleEvent(event);
|
|
440
|
+
}
|
|
441
|
+
/** Dispatches a typed xAI realtime server event to the appropriate behavior. */
|
|
442
|
+
handleEvent(event) {
|
|
443
|
+
switch (event.type) {
|
|
444
|
+
case 'response.output_audio_transcript.delta':
|
|
445
|
+
case 'response.audio_transcript.delta':
|
|
446
|
+
this.onAssistantDelta(event.delta);
|
|
447
|
+
break;
|
|
448
|
+
case 'response.output_audio_transcript.done':
|
|
449
|
+
case 'response.audio_transcript.done':
|
|
450
|
+
this.onAssistantDone(event.transcript);
|
|
451
|
+
break;
|
|
452
|
+
case 'response.audio.delta':
|
|
453
|
+
case 'response.output_audio.delta':
|
|
454
|
+
this.onAudioDelta(event.delta);
|
|
455
|
+
break;
|
|
456
|
+
case 'conversation.item.input_audio_transcription.completed':
|
|
457
|
+
this.onUserTranscript(event.transcript);
|
|
458
|
+
break;
|
|
459
|
+
case 'response.function_call_arguments.done':
|
|
460
|
+
this.onToolCallFrame(event);
|
|
461
|
+
break;
|
|
462
|
+
case 'input_audio_buffer.speech_started':
|
|
463
|
+
this.onSpeechStarted();
|
|
464
|
+
break;
|
|
465
|
+
case 'response.created':
|
|
466
|
+
this.responseActive = true;
|
|
467
|
+
// Stamp the kind of THIS response: 'narration' only when the flag was set by
|
|
468
|
+
// RequestSpokenUpdate immediately before its response.create (consumed here).
|
|
469
|
+
this.activeResponseKind = this.pendingNarrationKind ? 'narration' : 'normal';
|
|
470
|
+
this.pendingNarrationKind = false;
|
|
471
|
+
break;
|
|
472
|
+
case 'response.done':
|
|
473
|
+
// A turn finished — release the lock and speak any queued tool result so the
|
|
474
|
+
// model always voices the answer when delegated work comes back. The
|
|
475
|
+
// transcript-done frame for this turn has already arrived (it precedes
|
|
476
|
+
// response.done), so it's safe to reset the response kind here.
|
|
477
|
+
this.responseActive = false;
|
|
478
|
+
this.activeResponseKind = 'normal';
|
|
479
|
+
this.emitResponseUsage(event);
|
|
480
|
+
this.flushPendingResultResponse();
|
|
481
|
+
if (this.currentState === 'speaking') {
|
|
482
|
+
this.setState('listening');
|
|
483
|
+
}
|
|
484
|
+
break;
|
|
485
|
+
case 'error':
|
|
486
|
+
this.onErrorFrame(event);
|
|
487
|
+
break;
|
|
488
|
+
default:
|
|
489
|
+
// Unhandled event types are expected (the provider emits many); no-op.
|
|
490
|
+
break;
|
|
491
|
+
}
|
|
492
|
+
}
|
|
493
|
+
/** Appends an assistant transcript delta, reflects `'speaking'`, and emits the delta. */
|
|
494
|
+
onAssistantDelta(delta) {
|
|
495
|
+
if (this.currentState !== 'speaking') {
|
|
496
|
+
this.setState('speaking');
|
|
497
|
+
}
|
|
498
|
+
this.pendingAssistantText += delta;
|
|
499
|
+
this.emitTranscript({ Role: 'Assistant', Text: delta, IsFinal: false, Kind: this.activeResponseKind });
|
|
500
|
+
}
|
|
501
|
+
/**
|
|
502
|
+
* Finalizes the assistant turn: emits the final transcript tagged with the ACTIVE response
|
|
503
|
+
* kind (the transcript-done frame arrives BEFORE `response.done`, so
|
|
504
|
+
* {@link activeResponseKind} still reflects this turn), then returns to `'listening'`. Empty
|
|
505
|
+
* turns emit nothing.
|
|
506
|
+
*/
|
|
507
|
+
onAssistantDone(transcript) {
|
|
508
|
+
const finalText = transcript || this.pendingAssistantText;
|
|
509
|
+
this.pendingAssistantText = '';
|
|
510
|
+
if (finalText.trim().length > 0) {
|
|
511
|
+
this.emitTranscript({ Role: 'Assistant', Text: finalText, IsFinal: true, Kind: this.activeResponseKind });
|
|
512
|
+
}
|
|
513
|
+
if (this.currentState === 'speaking') {
|
|
514
|
+
this.setState('listening');
|
|
515
|
+
}
|
|
516
|
+
}
|
|
517
|
+
/**
|
|
518
|
+
* Decodes one base64 PCM16 chunk of the agent's spoken output into the playout queue and
|
|
519
|
+
* reflects `'speaking'`. The client OWNS the audio plane on this websocket transport (unlike
|
|
520
|
+
* the WebRTC OpenAI driver where the peer connection plays the remote track), so agent audio
|
|
521
|
+
* arrives as these deltas and is scheduled by the shared playout engine.
|
|
522
|
+
*/
|
|
523
|
+
onAudioDelta(deltaBase64) {
|
|
524
|
+
if (!deltaBase64) {
|
|
525
|
+
return;
|
|
526
|
+
}
|
|
527
|
+
if (this.currentState !== 'speaking') {
|
|
528
|
+
this.setState('speaking');
|
|
529
|
+
}
|
|
530
|
+
this.playback?.Enqueue(base64ToArrayBuffer(deltaBase64));
|
|
531
|
+
}
|
|
532
|
+
/**
|
|
533
|
+
* Emits the user's spoken-input transcription. Grok STREAMS this — repeated
|
|
534
|
+
* `input_audio_transcription.completed` events, each carrying the full text so far — unlike OpenAI's
|
|
535
|
+
* single final. So the FIRST emission of a turn appends a fresh caption and every later one is flagged
|
|
536
|
+
* ReplacesPrevious, collapsing the stream into ONE in-place-updating user bubble. The per-turn flag
|
|
537
|
+
* resets on the next `speech_started` ({@link onSpeechStarted}).
|
|
538
|
+
*/
|
|
539
|
+
onUserTranscript(transcript) {
|
|
540
|
+
if (transcript && transcript.trim().length > 0) {
|
|
541
|
+
this.emitTranscript({
|
|
542
|
+
Role: 'User', Text: transcript, IsFinal: true, Kind: 'normal',
|
|
543
|
+
ReplacesPrevious: this.userTurnTranscribed,
|
|
544
|
+
});
|
|
545
|
+
this.userTurnTranscribed = true;
|
|
546
|
+
}
|
|
547
|
+
}
|
|
548
|
+
/**
|
|
549
|
+
* Surfaces a completed tool call to the host. Two deliberate behaviors mirror the OpenAI
|
|
550
|
+
* driver: (1) the client silently leaves `'speaking'` (NO emission) so a host-rendered busy
|
|
551
|
+
* indicator isn't clobbered by this turn's trailing `response.done` / playback frames
|
|
552
|
+
* (obligation #1); (2) {@link responseActive} is CLEARED — the model has yielded the floor
|
|
553
|
+
* pending the result, so a queued {@link SendToolResult} can never deadlock (obligation #2).
|
|
554
|
+
*/
|
|
555
|
+
onToolCallFrame(call) {
|
|
556
|
+
if (this.currentState === 'speaking') {
|
|
557
|
+
this.currentState = 'connected';
|
|
558
|
+
}
|
|
559
|
+
this.responseActive = false;
|
|
560
|
+
this.emitToolCall({ CallID: call.call_id, ToolName: call.name, ArgumentsJson: call.arguments });
|
|
561
|
+
}
|
|
562
|
+
/**
|
|
563
|
+
* The user started speaking. This is a TRUE barge-in only when it cut off active model output
|
|
564
|
+
* (a response in flight or audio audibly playing) — a normal turn while the model is idle is
|
|
565
|
+
* NOT an interruption, so the emission is gated (base-contract rule). On a true barge-in the
|
|
566
|
+
* client OWNS the audio plane, so it flushes its own playout queue NOW (obligation #3),
|
|
567
|
+
* surfaces the interruption, and returns the floor; the provider cancels its own turn and
|
|
568
|
+
* emits a terminal `response.done`.
|
|
569
|
+
*/
|
|
570
|
+
onSpeechStarted() {
|
|
571
|
+
// A new user turn begins: the next transcription emission appends a fresh caption, and the rest
|
|
572
|
+
// of this turn's streamed transcriptions replace it in place (see onUserTranscript).
|
|
573
|
+
this.userTurnTranscribed = false;
|
|
574
|
+
if (!this.responseActive && !this.IsAudioPlaying) {
|
|
575
|
+
this.setState('listening');
|
|
576
|
+
return;
|
|
577
|
+
}
|
|
578
|
+
this.playback?.Flush();
|
|
579
|
+
this.responseActive = false;
|
|
580
|
+
this.activeResponseKind = 'normal';
|
|
581
|
+
this.emitInterruption();
|
|
582
|
+
this.setState('listening');
|
|
583
|
+
}
|
|
584
|
+
/**
|
|
585
|
+
* Emits the completed response's usage to the host as a DELTA (the `response.done` usage
|
|
586
|
+
* payload covers exactly this response, so it is already incremental — the `OnUsage`
|
|
587
|
+
* contract's preferred shape). Frames without a usage payload emit nothing.
|
|
588
|
+
*/
|
|
589
|
+
emitResponseUsage(event) {
|
|
590
|
+
const usage = event.response?.usage;
|
|
591
|
+
if (!usage) {
|
|
592
|
+
return;
|
|
593
|
+
}
|
|
594
|
+
this.emitUsage({
|
|
595
|
+
InputTokens: typeof usage.input_tokens === 'number' ? usage.input_tokens : undefined,
|
|
596
|
+
OutputTokens: typeof usage.output_tokens === 'number' ? usage.output_tokens : undefined,
|
|
597
|
+
Raw: usage,
|
|
598
|
+
});
|
|
599
|
+
}
|
|
600
|
+
/** Surfaces a provider error frame (non-fatal; the session continues). */
|
|
601
|
+
onErrorFrame(event) {
|
|
602
|
+
this.emitError({
|
|
603
|
+
Message: event.error?.message ?? 'Unknown provider error',
|
|
604
|
+
Code: event.error?.code,
|
|
605
|
+
Fatal: false,
|
|
606
|
+
});
|
|
607
|
+
}
|
|
608
|
+
// ── Response state machine ─────────────────────────────────────────────────
|
|
609
|
+
/**
|
|
610
|
+
* Asks the model to speak (a tool result or typed-text reply) — immediately if it's idle,
|
|
611
|
+
* otherwise queued until the current response finishes. An immediate trigger also CONSUMES
|
|
612
|
+
* any queued trigger debt: every payload item is already in the conversation, so one
|
|
613
|
+
* `response.create` voices everything (e.g. typed text barging in over a narration that had
|
|
614
|
+
* tool results queued behind it).
|
|
615
|
+
*/
|
|
616
|
+
requestResultResponse() {
|
|
617
|
+
if (!this.socket) {
|
|
618
|
+
return;
|
|
619
|
+
}
|
|
620
|
+
if (this.responseActive) {
|
|
621
|
+
this.pendingResultResponse = true;
|
|
622
|
+
return;
|
|
623
|
+
}
|
|
624
|
+
this.pendingResultResponse = false;
|
|
625
|
+
this.responseActive = true;
|
|
626
|
+
this.sendEvent({ type: 'response.create' });
|
|
627
|
+
this.setState('speaking');
|
|
628
|
+
}
|
|
629
|
+
/** On a turn completing, fire any queued tool-result response so the answer is spoken. */
|
|
630
|
+
flushPendingResultResponse() {
|
|
631
|
+
if (!this.pendingResultResponse || !this.socket) {
|
|
632
|
+
return;
|
|
633
|
+
}
|
|
634
|
+
this.pendingResultResponse = false;
|
|
635
|
+
this.responseActive = true;
|
|
636
|
+
this.sendEvent({ type: 'response.create' });
|
|
637
|
+
this.setState('speaking');
|
|
638
|
+
}
|
|
639
|
+
/** Resets the per-session response state machine (used on Disconnect). */
|
|
640
|
+
resetResponseState() {
|
|
641
|
+
this.pendingAssistantText = '';
|
|
642
|
+
this.responseActive = false;
|
|
643
|
+
this.pendingResultResponse = false;
|
|
644
|
+
this.pendingNarrationKind = false;
|
|
645
|
+
this.activeResponseKind = 'normal';
|
|
646
|
+
}
|
|
647
|
+
// ── Helpers ────────────────────────────────────────────────────────────────
|
|
648
|
+
/** Updates the client's own state view and emits the change to the host. */
|
|
649
|
+
setState(state) {
|
|
650
|
+
this.currentState = state;
|
|
651
|
+
this.emitStateChange(state);
|
|
652
|
+
}
|
|
653
|
+
/** JSON-serializes + sends a client event over the socket (only when open). */
|
|
654
|
+
sendEvent(event) {
|
|
655
|
+
// DIAGNOSTIC: log outbound control frames (skip the high-frequency mic audio) so we can see
|
|
656
|
+
// session.update / conversation.item.create / response.create actually went out — the other
|
|
657
|
+
// half of diagnosing a silent session.
|
|
658
|
+
if (event.type !== 'input_audio_buffer.append') {
|
|
659
|
+
console.debug('[xAIRealtimeClient] ▶ outbound:', event.type);
|
|
660
|
+
}
|
|
661
|
+
this.socket?.send(JSON.stringify(event));
|
|
662
|
+
}
|
|
663
|
+
};
|
|
664
|
+
xAIRealtimeClient = __decorate([
|
|
665
|
+
RegisterClass(BaseRealtimeClient, 'xai')
|
|
666
|
+
], xAIRealtimeClient);
|
|
667
|
+
export { xAIRealtimeClient };
|
|
668
|
+
/**
|
|
669
|
+
* Tree-shaking prevention: bundlers cannot see that {@link xAIRealtimeClient} is instantiated
|
|
670
|
+
* dynamically through the ClassFactory, so a consumer must call this no-op to create a static
|
|
671
|
+
* code path that keeps the `@RegisterClass` side effect alive.
|
|
672
|
+
*/
|
|
673
|
+
export function LoadxAIRealtimeClient() {
|
|
674
|
+
// intentional no-op — the static import of this module is the point
|
|
675
|
+
}
|
|
676
|
+
//# sourceMappingURL=xaiRealtimeClient.js.map
|