@memberjunction/ai-realtime-client 0.0.1 → 5.42.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +148 -28
- package/dist/audio/audioMeter.d.ts +101 -0
- package/dist/audio/audioMeter.d.ts.map +1 -0
- package/dist/audio/audioMeter.js +193 -0
- package/dist/audio/audioMeter.js.map +1 -0
- package/dist/audio/micCapture.d.ts +26 -0
- package/dist/audio/micCapture.d.ts.map +1 -0
- package/dist/audio/micCapture.js +69 -0
- package/dist/audio/micCapture.js.map +1 -0
- package/dist/audio/pcmPlayback.d.ts +73 -0
- package/dist/audio/pcmPlayback.d.ts.map +1 -0
- package/dist/audio/pcmPlayback.js +78 -0
- package/dist/audio/pcmPlayback.js.map +1 -0
- package/dist/audio/pcmUtils.d.ts +17 -0
- package/dist/audio/pcmUtils.d.ts.map +1 -0
- package/dist/audio/pcmUtils.js +46 -0
- package/dist/audio/pcmUtils.js.map +1 -0
- package/dist/drivers/assemblyAIRealtimeClient.d.ts +384 -0
- package/dist/drivers/assemblyAIRealtimeClient.d.ts.map +1 -0
- package/dist/drivers/assemblyAIRealtimeClient.js +732 -0
- package/dist/drivers/assemblyAIRealtimeClient.js.map +1 -0
- package/dist/drivers/elevenLabsRealtimeClient.d.ts +362 -0
- package/dist/drivers/elevenLabsRealtimeClient.d.ts.map +1 -0
- package/dist/drivers/elevenLabsRealtimeClient.js +686 -0
- package/dist/drivers/elevenLabsRealtimeClient.js.map +1 -0
- package/dist/drivers/geminiRealtimeClient.d.ts +406 -0
- package/dist/drivers/geminiRealtimeClient.d.ts.map +1 -0
- package/dist/drivers/geminiRealtimeClient.js +675 -0
- package/dist/drivers/geminiRealtimeClient.js.map +1 -0
- package/dist/drivers/openAIRealtimeClient.d.ts +381 -0
- package/dist/drivers/openAIRealtimeClient.d.ts.map +1 -0
- package/dist/drivers/openAIRealtimeClient.js +602 -0
- package/dist/drivers/openAIRealtimeClient.js.map +1 -0
- package/dist/drivers/xaiRealtimeClient.d.ts +430 -0
- package/dist/drivers/xaiRealtimeClient.d.ts.map +1 -0
- package/dist/drivers/xaiRealtimeClient.js +676 -0
- package/dist/drivers/xaiRealtimeClient.js.map +1 -0
- package/dist/generic/baseRealtimeClient.d.ts +401 -0
- package/dist/generic/baseRealtimeClient.d.ts.map +1 -0
- package/dist/generic/baseRealtimeClient.js +226 -0
- package/dist/generic/baseRealtimeClient.js.map +1 -0
- package/dist/index.d.ts +11 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11 -0
- package/dist/index.js.map +1 -0
- package/package.json +28 -7
|
@@ -0,0 +1,686 @@
|
|
|
1
|
+
var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
|
|
2
|
+
var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
|
|
3
|
+
if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
|
|
4
|
+
else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
|
|
5
|
+
return c > 3 && r && Object.defineProperty(target, key, r), r;
|
|
6
|
+
};
|
|
7
|
+
var ElevenLabsRealtimeClient_1;
|
|
8
|
+
import { RegisterClass } from '@memberjunction/global';
|
|
9
|
+
import { BaseRealtimeClient } from '../generic/baseRealtimeClient.js';
|
|
10
|
+
import { base64ToArrayBuffer } from '../audio/pcmUtils.js';
|
|
11
|
+
import { RealtimePcmPlayback } from '../audio/pcmPlayback.js';
|
|
12
|
+
import { RealtimeAudioMeter } from '../audio/audioMeter.js';
|
|
13
|
+
import { createPcmMicCapture } from '../audio/micCapture.js';
|
|
14
|
+
// ── Audio constants (ElevenLabs Agents wire formats) ───────────────────────────
|
|
15
|
+
/**
|
|
16
|
+
* Fallback PCM16 sample rate used when the `conversation_initiation_metadata` carries no
|
|
17
|
+
* parseable `pcm_<rate>` format (the platform default for both directions is `pcm_16000`).
|
|
18
|
+
*/
|
|
19
|
+
const ELEVENLABS_DEFAULT_SAMPLE_RATE = 16000;
|
|
20
|
+
// ── The driver ─────────────────────────────────────────────────────────────────
|
|
21
|
+
/**
|
|
22
|
+
* ElevenLabs Agents implementation of {@link BaseRealtimeClient}: a **browser-direct**
|
|
23
|
+
* conversation websocket authenticated with the server-minted SIGNED URL (the
|
|
24
|
+
* `EphemeralToken` IS the `wss://…&token=…` URL — no API key ever reaches the browser).
|
|
25
|
+
*
|
|
26
|
+
* Registered with the ClassFactory under the key `'elevenlabs'` — the `Provider` string the
|
|
27
|
+
* server's `ElevenLabsRealtime` driver stamps on its `ClientRealtimeSessionConfig`.
|
|
28
|
+
*
|
|
29
|
+
* Owns ALL ElevenLabs wire concerns (the behavioral twin of the Gemini client driver — the
|
|
30
|
+
* audio plane is client-owned over a websocket, no WebRTC):
|
|
31
|
+
* - **Connect handshake**: open the signed-URL socket → send
|
|
32
|
+
* `conversation_initiation_client_data` carrying the server-authored prompt override (from
|
|
33
|
+
* the `SessionConfig` pact) → wait for `conversation_initiation_metadata` → negotiate the
|
|
34
|
+
* PCM rates from `user_input_audio_format` / `agent_output_audio_format` → build the audio
|
|
35
|
+
* plane → `'listening'`. The state is gated on the metadata (driver obligation #7): only
|
|
36
|
+
* then has the platform confirmed the session — including the per-session system prompt —
|
|
37
|
+
* is applied.
|
|
38
|
+
* - **Audio in**: mic PCM16 at the negotiated input rate via the shared
|
|
39
|
+
* {@link createPcmMicCapture} worklet pipeline, streamed as base64 `user_audio_chunk`
|
|
40
|
+
* frames (note: a bare-key frame, not a `type`d one — that is the wire contract).
|
|
41
|
+
* - **Audio out**: `audio` events decoded into the shared {@link RealtimePcmPlayback} at the
|
|
42
|
+
* negotiated output rate; {@link IsAudioPlaying} comes from its playout clock.
|
|
43
|
+
* - **Transcripts are FINAL-only**: the Agents socket emits whole-utterance
|
|
44
|
+
* `user_transcript` / `agent_response` events (no interim deltas), so every transcript this
|
|
45
|
+
* driver emits has `IsFinal: true`. `agent_response_correction` (post-barge-in truncation)
|
|
46
|
+
* re-finalizes the assistant turn with the text that was ACTUALLY spoken — hosts persisting
|
|
47
|
+
* transcripts should treat a final assistant transcript arriving right after an
|
|
48
|
+
* interruption as the authoritative replacement of the previous one.
|
|
49
|
+
* - **Busy mapping**: `IsBusy` is set on the first `audio` / `agent_response` of a turn (and
|
|
50
|
+
* eagerly when this client triggers a response) and cleared on `agent_response_complete`,
|
|
51
|
+
* on `interruption`, and on a `client_tool_call` frame (the agent has yielded the floor
|
|
52
|
+
* pending the result — deadlock guard, obligation #2).
|
|
53
|
+
* - **Narration**: ElevenLabs has no per-response-instructions channel, so
|
|
54
|
+
* {@link RequestSpokenUpdate} is EMULATED Gemini-style — the instructions ride as a
|
|
55
|
+
* `user_message` and the response kind is stamped `'narration'` AT SEND TIME (sends are the
|
|
56
|
+
* turn triggers; there is no `response.created`-style confirmation frame to stamp on), then
|
|
57
|
+
* reset on the response boundary. Fidelity caveat: the instruction enters the conversation
|
|
58
|
+
* as a user turn, so the agent may occasionally reference it; hosts should phrase narration
|
|
59
|
+
* instructions accordingly.
|
|
60
|
+
* - **Cancel**: the protocol has no server-side cancel frame. {@link CancelActiveResponse}
|
|
61
|
+
* flushes the locally-owned playout queue (speech stops immediately) and marks the response
|
|
62
|
+
* inactive; residual server-side generation for a cancelled turn is simply not played.
|
|
63
|
+
* Spoken barge-in is handled by the platform's own VAD (which emits `interruption`), and a
|
|
64
|
+
* typed {@link SendText} takes the floor server-side as a fresh user turn.
|
|
65
|
+
* - **No usage telemetry**: the Conversational AI socket exposes no token-usage events, so
|
|
66
|
+
* this driver NEVER emits {@link OnUsage} (registering a handler is safe; it just never
|
|
67
|
+
* fires — usage accounting for ElevenLabs sessions happens platform-side).
|
|
68
|
+
*/
|
|
69
|
+
let ElevenLabsRealtimeClient = class ElevenLabsRealtimeClient extends BaseRealtimeClient {
|
|
70
|
+
constructor() {
|
|
71
|
+
super(...arguments);
|
|
72
|
+
// ── Transport / audio resources ────────────────────────────────────────────
|
|
73
|
+
this.socket = null;
|
|
74
|
+
this.micStream = null;
|
|
75
|
+
this.micCapture = null;
|
|
76
|
+
this.playback = null;
|
|
77
|
+
// ── Response state machine ─────────────────────────────────────────────────
|
|
78
|
+
/** True while an agent response is in flight; gates (queues) client-triggered sends. */
|
|
79
|
+
this.responseActive = false;
|
|
80
|
+
/** The kind of the response currently in flight; stamped at send time, reset on boundary. */
|
|
81
|
+
this.activeResponseKind = 'normal';
|
|
82
|
+
/** Sends deferred while a response is in flight; drained in order at the next boundary. */
|
|
83
|
+
this.queuedSends = [];
|
|
84
|
+
/**
|
|
85
|
+
* Nudge guard for tool results delivered AFTER the platform closed the turn (a long
|
|
86
|
+
* delegation outlasting the agent's last utterance): {@link sendToolResultFrame}
|
|
87
|
+
* optimistically marks a response active, but if NO model output actually arrives within
|
|
88
|
+
* this window, the platform has silently absorbed the result — the timer fires a
|
|
89
|
+
* user_message nudge so the outcome is always voiced (live finding).
|
|
90
|
+
*/
|
|
91
|
+
this.toolResultNudgeTimer = null;
|
|
92
|
+
/**
|
|
93
|
+
* Tool calls awaiting their result — used to make {@link SendToolResult} EXACTLY-ONCE:
|
|
94
|
+
* the id is consumed when the result is accepted (queued or sent), so a duplicate send
|
|
95
|
+
* for the same call is dropped with a warning instead of confusing the conversation.
|
|
96
|
+
*/
|
|
97
|
+
this.pendingToolCallIds = new Set();
|
|
98
|
+
/** True once Disconnect ran — an expected socket close must not surface as fatal. */
|
|
99
|
+
this.closedByConsumer = false;
|
|
100
|
+
/**
|
|
101
|
+
* The client's own view of the session state — mirrors what was last emitted, EXCEPT after
|
|
102
|
+
* a tool call: the host typically shows its own busy indicator then, so the client silently
|
|
103
|
+
* leaves `'speaking'` (no emission) until the result reply's first output re-asserts it.
|
|
104
|
+
*/
|
|
105
|
+
this.currentState = 'closed';
|
|
106
|
+
// ── Connection internals ───────────────────────────────────────────────────
|
|
107
|
+
/** Resolver for the in-flight Connect's metadata wait (null outside Connect). */
|
|
108
|
+
this.onInitiationMetadata = null;
|
|
109
|
+
}
|
|
110
|
+
static { ElevenLabsRealtimeClient_1 = this; }
|
|
111
|
+
// ── BaseRealtimeClient: connection lifecycle ───────────────────────────────
|
|
112
|
+
/**
|
|
113
|
+
* Opens the client-direct conversation: socket to the signed URL, the initiation frame
|
|
114
|
+
* carrying the server-authored prompt override, then — once the metadata confirms the
|
|
115
|
+
* session config is applied — the audio plane at the NEGOTIATED sample rates. Reports
|
|
116
|
+
* `'listening'` only after all of that (obligation #7).
|
|
117
|
+
*/
|
|
118
|
+
async Connect(config, micStream) {
|
|
119
|
+
this.micStream = micStream;
|
|
120
|
+
this.closedByConsumer = false;
|
|
121
|
+
this.setState('connecting');
|
|
122
|
+
const overrides = this.parseOverrides(config);
|
|
123
|
+
let openSocket = null;
|
|
124
|
+
let failOpen = null;
|
|
125
|
+
let confirmMetadata = null;
|
|
126
|
+
let failMetadata = null;
|
|
127
|
+
const opened = new Promise((resolve, reject) => {
|
|
128
|
+
openSocket = resolve;
|
|
129
|
+
failOpen = reject;
|
|
130
|
+
});
|
|
131
|
+
const metadata = new Promise((resolve, reject) => {
|
|
132
|
+
confirmMetadata = resolve;
|
|
133
|
+
failMetadata = reject;
|
|
134
|
+
});
|
|
135
|
+
// When the socket dies BEFORE open, Connect throws at `await opened` and never awaits
|
|
136
|
+
// `metadata` — pre-consume its rejection so it can't surface as an unhandled rejection.
|
|
137
|
+
metadata.catch(() => undefined);
|
|
138
|
+
this.onInitiationMetadata = confirmMetadata;
|
|
139
|
+
const failConnect = (error) => {
|
|
140
|
+
failOpen?.(error);
|
|
141
|
+
failMetadata?.(error);
|
|
142
|
+
};
|
|
143
|
+
const socket = this.createSocket(config.EphemeralToken);
|
|
144
|
+
this.socket = socket;
|
|
145
|
+
socket.onopen = () => openSocket?.();
|
|
146
|
+
socket.onmessage = (data) => this.handleSocketMessage(data);
|
|
147
|
+
socket.onerror = (message) => {
|
|
148
|
+
failConnect(new Error(message));
|
|
149
|
+
this.handleSocketError(message);
|
|
150
|
+
};
|
|
151
|
+
socket.onclose = () => {
|
|
152
|
+
failConnect(new Error('ElevenLabs conversation socket closed during connect'));
|
|
153
|
+
this.handleSocketClose();
|
|
154
|
+
};
|
|
155
|
+
await opened;
|
|
156
|
+
this.setState('connected');
|
|
157
|
+
// The server-authored prompt override (the SessionConfig pact) is applied via the
|
|
158
|
+
// initiation frame — the managed agent's platform settings enable exactly this field.
|
|
159
|
+
this.sendFrame({ type: 'conversation_initiation_client_data', conversation_config_override: overrides });
|
|
160
|
+
const metadataEvent = await metadata;
|
|
161
|
+
const formats = metadataEvent.conversation_initiation_metadata_event;
|
|
162
|
+
const outputRate = ElevenLabsRealtimeClient_1.ParsePcmRate(formats?.agent_output_audio_format, 'output');
|
|
163
|
+
const inputRate = ElevenLabsRealtimeClient_1.ParsePcmRate(formats?.user_input_audio_format, 'input');
|
|
164
|
+
this.playback = this.createPlayback(outputRate);
|
|
165
|
+
this.micCapture = await this.createMicCapture(micStream, inputRate, (base64Pcm16) => this.sendMicChunk(base64Pcm16));
|
|
166
|
+
// Audio-activity capability (base obligation #9): agent side taps the playout
|
|
167
|
+
// engine's master gain; user side meters the mic stream. Null-safe — test fakes /
|
|
168
|
+
// no-WebAudio environments simply leave the session un-metered.
|
|
169
|
+
this.attachOutputAudioMeter(this.playback?.CreateMeter?.() ?? null);
|
|
170
|
+
this.attachInputAudioMeter(RealtimeAudioMeter.ForMicStream(micStream));
|
|
171
|
+
this.setState('listening');
|
|
172
|
+
}
|
|
173
|
+
/**
|
|
174
|
+
* Tears down the socket, mic capture, mic tracks, and playout engine, resets the response
|
|
175
|
+
* state machine, and emits a final `'closed'` (unless already `'error'`). Safe to call
|
|
176
|
+
* more than once.
|
|
177
|
+
*/
|
|
178
|
+
async Disconnect() {
|
|
179
|
+
this.closedByConsumer = true;
|
|
180
|
+
this.closeAudioMeters();
|
|
181
|
+
this.micStream?.getTracks().forEach((track) => track.stop());
|
|
182
|
+
this.micStream = null;
|
|
183
|
+
this.micCapture?.Stop();
|
|
184
|
+
this.micCapture = null;
|
|
185
|
+
this.playback?.Close();
|
|
186
|
+
this.playback = null;
|
|
187
|
+
if (this.socket) {
|
|
188
|
+
try {
|
|
189
|
+
this.socket.close();
|
|
190
|
+
}
|
|
191
|
+
catch {
|
|
192
|
+
/* already closing */
|
|
193
|
+
}
|
|
194
|
+
this.socket = null;
|
|
195
|
+
}
|
|
196
|
+
this.resetResponseState();
|
|
197
|
+
if (this.currentState !== 'error') {
|
|
198
|
+
this.setState('closed');
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
// ── BaseRealtimeClient: outbound actions ──────────────────────────────────
|
|
202
|
+
/**
|
|
203
|
+
* Injects typed text as a `user_message` (the platform's "respond to this" trigger).
|
|
204
|
+
* No-op when the session is not open.
|
|
205
|
+
*
|
|
206
|
+
* **SendText implies barge-in** (base-contract rule): an active spoken response is
|
|
207
|
+
* cancelled via {@link CancelActiveResponse} first — playback flushed, response marked
|
|
208
|
+
* inactive, queued sends drained — so the typed turn takes the floor immediately. If a
|
|
209
|
+
* drained queued send (e.g. a pending tool result) starts a new response, the text queues
|
|
210
|
+
* behind it, preserving the tool-result delivery invariant. Server-side the fresh user
|
|
211
|
+
* turn takes the floor on its own; no explicit cancel frame exists or is needed.
|
|
212
|
+
*/
|
|
213
|
+
SendText(text) {
|
|
214
|
+
if (!this.socket) {
|
|
215
|
+
return;
|
|
216
|
+
}
|
|
217
|
+
this.CancelActiveResponse();
|
|
218
|
+
this.enqueueOrRun(() => this.sendUserMessage(text, 'normal', true));
|
|
219
|
+
}
|
|
220
|
+
/**
|
|
221
|
+
* @inheritdoc
|
|
222
|
+
*
|
|
223
|
+
* ElevenLabs has no cancel frame — the client OWNS the audio plane, so cancelling means:
|
|
224
|
+
* flush the local playout queue (speech stops immediately), mark the in-flight response
|
|
225
|
+
* inactive, and drain queued sends (a queued tool result takes the floor next — delivery
|
|
226
|
+
* is never dropped by a cancel). Residual server-side generation for the cancelled turn is
|
|
227
|
+
* simply never played; the platform's own VAD handles SPOKEN barge-in (emitting
|
|
228
|
+
* `interruption`). No-op when nothing is active.
|
|
229
|
+
*/
|
|
230
|
+
CancelActiveResponse() {
|
|
231
|
+
if (!this.socket) {
|
|
232
|
+
return;
|
|
233
|
+
}
|
|
234
|
+
if (!this.responseActive && !this.IsAudioPlaying) {
|
|
235
|
+
return; // nothing active — no-op by contract
|
|
236
|
+
}
|
|
237
|
+
this.playback?.Flush();
|
|
238
|
+
this.responseActive = false;
|
|
239
|
+
this.activeResponseKind = 'normal';
|
|
240
|
+
this.flushQueuedSends();
|
|
241
|
+
if (this.currentState === 'speaking') {
|
|
242
|
+
this.setState('listening');
|
|
243
|
+
}
|
|
244
|
+
}
|
|
245
|
+
/**
|
|
246
|
+
* Injects background context via `contextual_update` — NATIVE on this provider: the
|
|
247
|
+
* platform's purpose-built non-interrupting context channel. Sent immediately even while
|
|
248
|
+
* a response is in flight (the platform guarantees it never triggers or disturbs
|
|
249
|
+
* generation), so unlike the Gemini driver no queueing applies.
|
|
250
|
+
*/
|
|
251
|
+
SendContextNote(text) {
|
|
252
|
+
if (!this.socket) {
|
|
253
|
+
return;
|
|
254
|
+
}
|
|
255
|
+
this.sendFrame({ type: 'contextual_update', text });
|
|
256
|
+
}
|
|
257
|
+
/**
|
|
258
|
+
* Triggers ONE short spoken update. EMULATED as a `user_message` carrying the
|
|
259
|
+
* instructions, with the resulting response stamped `Kind: 'narration'` at send time
|
|
260
|
+
* (reset on the response boundary) — see the class-level fidelity caveat. Queued behind
|
|
261
|
+
* any in-flight response (a `user_message` sent mid-response would barge in on it), per
|
|
262
|
+
* the base contract's collision rule.
|
|
263
|
+
*/
|
|
264
|
+
RequestSpokenUpdate(instructions) {
|
|
265
|
+
if (!this.socket) {
|
|
266
|
+
return;
|
|
267
|
+
}
|
|
268
|
+
this.enqueueOrRun(() => this.sendUserMessage(instructions, 'narration', false));
|
|
269
|
+
}
|
|
270
|
+
/**
|
|
271
|
+
* Feeds an executed tool's result back via `client_tool_result`, correlated by the
|
|
272
|
+
* platform's `tool_call_id`. EXACTLY-ONCE: the pending id is consumed when the result is
|
|
273
|
+
* accepted, and a duplicate (or unknown) callID is dropped with a warning. Sent
|
|
274
|
+
* immediately when idle — the platform speaks the result as the turn's continuation —
|
|
275
|
+
* otherwise queued behind the in-flight response (e.g. a progress narration) so the
|
|
276
|
+
* trigger is never lost.
|
|
277
|
+
*/
|
|
278
|
+
SendToolResult(callID, outputJson) {
|
|
279
|
+
if (!this.socket) {
|
|
280
|
+
return;
|
|
281
|
+
}
|
|
282
|
+
if (!this.pendingToolCallIds.has(callID)) {
|
|
283
|
+
console.warn(`ElevenLabsRealtimeClient.SendToolResult: no pending tool call '${callID}' — duplicate or unknown result dropped.`);
|
|
284
|
+
return;
|
|
285
|
+
}
|
|
286
|
+
this.pendingToolCallIds.delete(callID);
|
|
287
|
+
this.enqueueOrRun(() => this.sendToolResultFrame(callID, outputJson));
|
|
288
|
+
}
|
|
289
|
+
/**
|
|
290
|
+
* Mutes / unmutes by toggling the mic tracks' `enabled` flag: the capture pipeline stays
|
|
291
|
+
* up and streams SILENCE while muted (the provider's VAD sees a continuous stream and the
|
|
292
|
+
* un-mute is glitch-free — same policy as the OpenAI and Gemini client drivers).
|
|
293
|
+
*/
|
|
294
|
+
SetMuted(muted) {
|
|
295
|
+
const tracks = this.micStream?.getAudioTracks() ?? [];
|
|
296
|
+
for (const track of tracks) {
|
|
297
|
+
track.enabled = !muted;
|
|
298
|
+
}
|
|
299
|
+
}
|
|
300
|
+
/** @inheritdoc */
|
|
301
|
+
get IsBusy() {
|
|
302
|
+
return this.responseActive;
|
|
303
|
+
}
|
|
304
|
+
/**
|
|
305
|
+
* @inheritdoc
|
|
306
|
+
*
|
|
307
|
+
* Computed directly from the playout engine's playhead clock — this client OWNS the
|
|
308
|
+
* output buffer, so "audibly playing" is precisely "scheduled audio extends beyond the
|
|
309
|
+
* audio context's current time".
|
|
310
|
+
*/
|
|
311
|
+
get IsAudioPlaying() {
|
|
312
|
+
return this.playback?.IsPlaying ?? false;
|
|
313
|
+
}
|
|
314
|
+
// ── Overridable creation seams (tests inject fakes — no network / audio) ──
|
|
315
|
+
/**
|
|
316
|
+
* Creation seam for the conversation websocket. Production wraps the platform-global
|
|
317
|
+
* `WebSocket` opened against the signed URL; unit tests override this to return an
|
|
318
|
+
* in-memory fake. Handlers are attached by {@link Connect} AFTER this returns, so the
|
|
319
|
+
* implementation must not require them at construction time.
|
|
320
|
+
*/
|
|
321
|
+
createSocket(signedUrl) {
|
|
322
|
+
const WS = globalThis.WebSocket;
|
|
323
|
+
if (!WS) {
|
|
324
|
+
throw new Error('ElevenLabsRealtimeClient requires a global WebSocket (browser or Node 22+).');
|
|
325
|
+
}
|
|
326
|
+
const ws = new WS(signedUrl);
|
|
327
|
+
const seam = {
|
|
328
|
+
onopen: null,
|
|
329
|
+
onmessage: null,
|
|
330
|
+
onerror: null,
|
|
331
|
+
onclose: null,
|
|
332
|
+
send: (data) => ws.send(data),
|
|
333
|
+
close: () => ws.close(),
|
|
334
|
+
};
|
|
335
|
+
ws.onopen = () => seam.onopen?.();
|
|
336
|
+
ws.onmessage = (event) => seam.onmessage?.(String(event.data));
|
|
337
|
+
ws.onerror = () => seam.onerror?.('ElevenLabs conversation websocket error');
|
|
338
|
+
ws.onclose = () => seam.onclose?.();
|
|
339
|
+
return seam;
|
|
340
|
+
}
|
|
341
|
+
/**
|
|
342
|
+
* Creation seam for the mic-capture pipeline at the NEGOTIATED input rate. Production
|
|
343
|
+
* delegates to the shared {@link createPcmMicCapture}; unit tests override this with a
|
|
344
|
+
* no-op fake (and may capture `onPcmChunk` to simulate mic frames).
|
|
345
|
+
*/
|
|
346
|
+
async createMicCapture(micStream, sampleRate, onPcmChunk) {
|
|
347
|
+
return createPcmMicCapture(micStream, sampleRate, onPcmChunk);
|
|
348
|
+
}
|
|
349
|
+
/**
|
|
350
|
+
* Creation seam for the playout engine at the NEGOTIATED output rate. Production returns
|
|
351
|
+
* the shared {@link RealtimePcmPlayback}.
|
|
352
|
+
*/
|
|
353
|
+
createPlayback(sampleRate) {
|
|
354
|
+
return new RealtimePcmPlayback(sampleRate);
|
|
355
|
+
}
|
|
356
|
+
/**
|
|
357
|
+
* Extracts the wire-shaped `conversation_config_override` from the server-minted
|
|
358
|
+
* `SessionConfig` pact (`{ agentId, overrides, config }` — authored by the server's
|
|
359
|
+
* `ElevenLabsRealtime.CreateClientSession`). Falls back to an empty override when absent
|
|
360
|
+
* (the managed agent then runs on its stored base prompt).
|
|
361
|
+
*/
|
|
362
|
+
parseOverrides(config) {
|
|
363
|
+
const sessionConfig = config.SessionConfig ?? {};
|
|
364
|
+
const raw = sessionConfig['overrides'];
|
|
365
|
+
return raw !== null && typeof raw === 'object' && !Array.isArray(raw) ? raw : {};
|
|
366
|
+
}
|
|
367
|
+
/**
|
|
368
|
+
* Parses a `pcm_<rate>` audio-format tag from the initiation metadata. Non-PCM formats
|
|
369
|
+
* (e.g. `ulaw_8000` — a telephony-only configuration) are not playable/encodable by the
|
|
370
|
+
* shared PCM pipeline; the driver warns and falls back to the platform default so the
|
|
371
|
+
* session degrades loudly rather than throwing.
|
|
372
|
+
*/
|
|
373
|
+
static ParsePcmRate(format, direction) {
|
|
374
|
+
const match = /^pcm_(\d+)$/.exec(format ?? '');
|
|
375
|
+
if (match) {
|
|
376
|
+
return Number(match[1]);
|
|
377
|
+
}
|
|
378
|
+
if (format) {
|
|
379
|
+
console.warn(`ElevenLabsRealtimeClient: unsupported ${direction} audio format '${format}' — only pcm_<rate> is ` +
|
|
380
|
+
`supported; falling back to pcm_${ELEVENLABS_DEFAULT_SAMPLE_RATE}. Configure the agent for PCM audio.`);
|
|
381
|
+
}
|
|
382
|
+
return ELEVENLABS_DEFAULT_SAMPLE_RATE;
|
|
383
|
+
}
|
|
384
|
+
/** Streams one base64 PCM16 mic chunk as a bare-key `user_audio_chunk` frame. */
|
|
385
|
+
sendMicChunk(base64Pcm16) {
|
|
386
|
+
if (this.socket) {
|
|
387
|
+
this.sendFrame({ user_audio_chunk: base64Pcm16 });
|
|
388
|
+
}
|
|
389
|
+
}
|
|
390
|
+
/** Surfaces a fatal socket error and marks the session unusable (obligation #6). */
|
|
391
|
+
handleSocketError(message) {
|
|
392
|
+
if (this.currentState === 'error' || this.currentState === 'closed') {
|
|
393
|
+
return;
|
|
394
|
+
}
|
|
395
|
+
this.emitError({ Message: `ElevenLabs realtime transport error: ${message}`, Fatal: true });
|
|
396
|
+
this.setState('error');
|
|
397
|
+
}
|
|
398
|
+
/**
|
|
399
|
+
* A socket close the CONSUMER didn't ask for is fatal: ElevenLabs hard-closes at signed-URL
|
|
400
|
+
* expiry and when the agent itself ends the conversation, so an unexpected close is how
|
|
401
|
+
* credential / conversation death reaches the host (obligation #6).
|
|
402
|
+
*/
|
|
403
|
+
handleSocketClose() {
|
|
404
|
+
if (this.closedByConsumer || this.currentState === 'error' || this.currentState === 'closed') {
|
|
405
|
+
return;
|
|
406
|
+
}
|
|
407
|
+
this.emitError({ Message: 'ElevenLabs conversation closed unexpectedly', Fatal: true });
|
|
408
|
+
this.setState('error');
|
|
409
|
+
}
|
|
410
|
+
// ── Inbound message translation ────────────────────────────────────────────
|
|
411
|
+
/** Parses one raw socket payload; non-JSON frames are ignored. */
|
|
412
|
+
handleSocketMessage(data) {
|
|
413
|
+
let event;
|
|
414
|
+
try {
|
|
415
|
+
event = JSON.parse(data);
|
|
416
|
+
}
|
|
417
|
+
catch {
|
|
418
|
+
return;
|
|
419
|
+
}
|
|
420
|
+
this.handleServerEvent(event);
|
|
421
|
+
}
|
|
422
|
+
/** Multiplexes one inbound frame to the focused per-concern handlers. */
|
|
423
|
+
handleServerEvent(event) {
|
|
424
|
+
switch (event.type) {
|
|
425
|
+
case 'conversation_initiation_metadata':
|
|
426
|
+
this.onInitiationMetadata?.(event);
|
|
427
|
+
this.onInitiationMetadata = null;
|
|
428
|
+
break;
|
|
429
|
+
case 'audio':
|
|
430
|
+
this.handleAudio(event.audio_event?.audio_base_64);
|
|
431
|
+
break;
|
|
432
|
+
case 'user_transcript':
|
|
433
|
+
this.handleUserTranscript(event.user_transcription_event?.user_transcript);
|
|
434
|
+
break;
|
|
435
|
+
case 'agent_response':
|
|
436
|
+
this.handleAgentResponse(event.agent_response_event?.agent_response);
|
|
437
|
+
break;
|
|
438
|
+
case 'agent_response_correction':
|
|
439
|
+
this.handleAgentResponseCorrection(event.agent_response_correction_event?.corrected_agent_response);
|
|
440
|
+
break;
|
|
441
|
+
case 'agent_response_complete':
|
|
442
|
+
this.handleResponseComplete();
|
|
443
|
+
break;
|
|
444
|
+
case 'client_tool_call':
|
|
445
|
+
this.handleClientToolCall(event.client_tool_call);
|
|
446
|
+
break;
|
|
447
|
+
case 'interruption':
|
|
448
|
+
this.handleInterruption();
|
|
449
|
+
break;
|
|
450
|
+
case 'ping':
|
|
451
|
+
this.sendFrame({ type: 'pong', event_id: event.ping_event?.event_id ?? 0 });
|
|
452
|
+
break;
|
|
453
|
+
case 'vad_score':
|
|
454
|
+
break; // continuous voice-activity telemetry — deliberately ignored
|
|
455
|
+
case 'guardrail_triggered':
|
|
456
|
+
console.warn('ElevenLabsRealtimeClient: agent guardrail triggered', event);
|
|
457
|
+
break;
|
|
458
|
+
default:
|
|
459
|
+
break; // unknown / future frame types are ignored
|
|
460
|
+
}
|
|
461
|
+
}
|
|
462
|
+
/** Decodes one base64 model-audio frame into the playout queue and marks generation live. */
|
|
463
|
+
handleAudio(audioBase64) {
|
|
464
|
+
if (!audioBase64) {
|
|
465
|
+
return;
|
|
466
|
+
}
|
|
467
|
+
this.markGenerationStarted();
|
|
468
|
+
this.playback?.Enqueue(base64ToArrayBuffer(audioBase64));
|
|
469
|
+
}
|
|
470
|
+
/** User transcript: whole-utterance FINAL (this provider emits no interim deltas). */
|
|
471
|
+
handleUserTranscript(text) {
|
|
472
|
+
if (text && text.trim().length > 0) {
|
|
473
|
+
this.emitTranscript({ Role: 'User', Text: text, IsFinal: true, Kind: 'normal' });
|
|
474
|
+
}
|
|
475
|
+
}
|
|
476
|
+
/**
|
|
477
|
+
* Agent response: the turn's complete text, emitted FINAL with the ACTIVE response kind so
|
|
478
|
+
* narration turns are tagged correctly (stamp-at-send — see the class doc).
|
|
479
|
+
*/
|
|
480
|
+
handleAgentResponse(text) {
|
|
481
|
+
this.markGenerationStarted();
|
|
482
|
+
if (text && text.trim().length > 0) {
|
|
483
|
+
this.emitTranscript({ Role: 'Assistant', Text: text, IsFinal: true, Kind: this.activeResponseKind });
|
|
484
|
+
}
|
|
485
|
+
}
|
|
486
|
+
/**
|
|
487
|
+
* Post-barge-in correction: re-finalizes the assistant turn with the text that was
|
|
488
|
+
* ACTUALLY spoken before the interruption cut it off. Emitted as a fresh FINAL assistant
|
|
489
|
+
* transcript, `Kind: 'normal'` (the interruption already reset the response kind; a
|
|
490
|
+
* truncated narration's correction is still just what was audibly said) — stamped
|
|
491
|
+
* {@link RealtimeClientTranscript.ReplacesPrevious} so hosts UPDATE the superseded turn
|
|
492
|
+
* in place instead of persisting both (closes the §10 "Transcript `Replaces` marker" gap).
|
|
493
|
+
*/
|
|
494
|
+
handleAgentResponseCorrection(correctedText) {
|
|
495
|
+
if (correctedText && correctedText.trim().length > 0) {
|
|
496
|
+
this.emitTranscript({ Role: 'Assistant', Text: correctedText, IsFinal: true, Kind: 'normal', ReplacesPrevious: true });
|
|
497
|
+
}
|
|
498
|
+
}
|
|
499
|
+
/**
|
|
500
|
+
* Response boundary: release the busy lock, reset the response kind, drain queued sends
|
|
501
|
+
* (stopping at the first one that starts a new response), and return the floor.
|
|
502
|
+
*/
|
|
503
|
+
handleResponseComplete() {
|
|
504
|
+
this.responseActive = false;
|
|
505
|
+
this.activeResponseKind = 'normal';
|
|
506
|
+
this.flushQueuedSends();
|
|
507
|
+
if (this.currentState === 'speaking') {
|
|
508
|
+
this.setState('listening');
|
|
509
|
+
}
|
|
510
|
+
}
|
|
511
|
+
/**
|
|
512
|
+
* Surfaces the agent's tool call to the host. Two deliberate behaviors mirror the other
|
|
513
|
+
* drivers: (1) the client silently leaves `'speaking'` (no emission) so a host-rendered
|
|
514
|
+
* busy indicator isn't clobbered by the turn's trailing frames (obligation #1); (2)
|
|
515
|
+
* `responseActive` is CLEARED — the agent has yielded the floor pending the result, so a
|
|
516
|
+
* queued send can never deadlock (obligation #2). The queue is NOT drained here: a queued
|
|
517
|
+
* narration must not inject a user turn between the tool call and its result.
|
|
518
|
+
*/
|
|
519
|
+
handleClientToolCall(call) {
|
|
520
|
+
if (!call) {
|
|
521
|
+
return;
|
|
522
|
+
}
|
|
523
|
+
if (this.currentState === 'speaking') {
|
|
524
|
+
this.currentState = 'connected';
|
|
525
|
+
}
|
|
526
|
+
this.responseActive = false;
|
|
527
|
+
const callID = call.tool_call_id ?? '';
|
|
528
|
+
this.pendingToolCallIds.add(callID);
|
|
529
|
+
this.emitToolCall({ CallID: callID, ToolName: call.tool_name ?? '', ArgumentsJson: JSON.stringify(call.parameters ?? {}) });
|
|
530
|
+
}
|
|
531
|
+
/**
|
|
532
|
+
* True barge-in: the platform's VAD detected the user cutting off active agent output
|
|
533
|
+
* (ElevenLabs only emits `interruption` for actual barge-ins, but the emission is still
|
|
534
|
+
* gated on something genuinely being active — a spurious frame while idle is NOT an
|
|
535
|
+
* interruption per the base contract). Flush playout, surface it, reset the response
|
|
536
|
+
* machine, give the floor back. The queue is NOT drained — the user has the floor; queued
|
|
537
|
+
* sends flush at the next response boundary.
|
|
538
|
+
*/
|
|
539
|
+
handleInterruption() {
|
|
540
|
+
this.cancelToolResultNudge();
|
|
541
|
+
if (!this.responseActive && !this.IsAudioPlaying) {
|
|
542
|
+
return;
|
|
543
|
+
}
|
|
544
|
+
this.playback?.Flush();
|
|
545
|
+
this.responseActive = false;
|
|
546
|
+
this.activeResponseKind = 'normal';
|
|
547
|
+
this.emitInterruption();
|
|
548
|
+
this.setState('listening');
|
|
549
|
+
}
|
|
550
|
+
/**
|
|
551
|
+
* First model output of a response (audio or the response text): the agent is busy and
|
|
552
|
+
* the client is audibly / imminently `'speaking'`.
|
|
553
|
+
*/
|
|
554
|
+
markGenerationStarted() {
|
|
555
|
+
this.cancelToolResultNudge();
|
|
556
|
+
this.responseActive = true;
|
|
557
|
+
if (this.currentState !== 'speaking') {
|
|
558
|
+
this.setState('speaking');
|
|
559
|
+
}
|
|
560
|
+
}
|
|
561
|
+
// ── Collision-safe send machinery ──────────────────────────────────────────
|
|
562
|
+
/**
|
|
563
|
+
* Runs a send immediately when no response is in flight; otherwise queues it for the next
|
|
564
|
+
* boundary (`agent_response_complete`). A `user_message` sent mid-response barges in on
|
|
565
|
+
* it, so deferral is the safe default — mirroring the Gemini driver's rule.
|
|
566
|
+
*/
|
|
567
|
+
enqueueOrRun(send) {
|
|
568
|
+
if (this.responseActive) {
|
|
569
|
+
this.queuedSends.push(send);
|
|
570
|
+
return;
|
|
571
|
+
}
|
|
572
|
+
send();
|
|
573
|
+
}
|
|
574
|
+
/**
|
|
575
|
+
* Drains queued sends in order at a response boundary, stopping as soon as one starts a
|
|
576
|
+
* new response (sets {@link responseActive}) — the rest wait for that response to finish.
|
|
577
|
+
*/
|
|
578
|
+
flushQueuedSends() {
|
|
579
|
+
while (!this.responseActive && this.queuedSends.length > 0) {
|
|
580
|
+
const send = this.queuedSends.shift();
|
|
581
|
+
send?.();
|
|
582
|
+
}
|
|
583
|
+
}
|
|
584
|
+
/**
|
|
585
|
+
* Sends a `user_message` that triggers a response, stamping the upcoming response's kind
|
|
586
|
+
* at send time and eagerly marking the agent busy. `emitSpeaking` mirrors the other
|
|
587
|
+
* drivers: typed text reflects `'speaking'` immediately; narration waits for the first
|
|
588
|
+
* model output.
|
|
589
|
+
*/
|
|
590
|
+
sendUserMessage(text, kind, emitSpeaking) {
|
|
591
|
+
if (!this.socket) {
|
|
592
|
+
return;
|
|
593
|
+
}
|
|
594
|
+
this.sendFrame({ type: 'user_message', text });
|
|
595
|
+
this.responseActive = true;
|
|
596
|
+
this.activeResponseKind = kind;
|
|
597
|
+
if (emitSpeaking) {
|
|
598
|
+
this.setState('speaking');
|
|
599
|
+
}
|
|
600
|
+
}
|
|
601
|
+
/**
|
|
602
|
+
* Sends the `client_tool_result` frame (the platform speaks the result as the turn's
|
|
603
|
+
* continuation — no explicit generation trigger exists or is needed) and eagerly marks
|
|
604
|
+
* the agent busy so a queued narration can't slip in before the spoken result.
|
|
605
|
+
*/
|
|
606
|
+
sendToolResultFrame(callID, outputJson) {
|
|
607
|
+
if (!this.socket) {
|
|
608
|
+
return;
|
|
609
|
+
}
|
|
610
|
+
this.sendFrame({
|
|
611
|
+
type: 'client_tool_result',
|
|
612
|
+
tool_call_id: callID,
|
|
613
|
+
result: ElevenLabsRealtimeClient_1.ParseToolOutput(outputJson),
|
|
614
|
+
is_error: false,
|
|
615
|
+
});
|
|
616
|
+
this.responseActive = true;
|
|
617
|
+
this.activeResponseKind = 'normal';
|
|
618
|
+
this.setState('speaking');
|
|
619
|
+
this.armToolResultNudge();
|
|
620
|
+
}
|
|
621
|
+
/** How long to wait for real model output after a tool result before nudging. */
|
|
622
|
+
static { this.ToolResultNudgeMs = 1600; }
|
|
623
|
+
/** Arms the absorbed-tool-result nudge (see {@link toolResultNudgeTimer}). */
|
|
624
|
+
armToolResultNudge() {
|
|
625
|
+
this.cancelToolResultNudge();
|
|
626
|
+
this.toolResultNudgeTimer = setTimeout(() => {
|
|
627
|
+
this.toolResultNudgeTimer = null;
|
|
628
|
+
// No audio / response text arrived — the platform closed the turn before the
|
|
629
|
+
// result landed and won't speak it on its own. Clear the optimistic busy mark
|
|
630
|
+
// and explicitly ask for the outcome.
|
|
631
|
+
this.responseActive = false;
|
|
632
|
+
this.sendUserMessage('The delegated work you were waiting on has just finished and its result has been ' +
|
|
633
|
+
'delivered to you. Tell the user the outcome now, in your own first-person voice.', 'normal', false);
|
|
634
|
+
}, ElevenLabsRealtimeClient_1.ToolResultNudgeMs);
|
|
635
|
+
}
|
|
636
|
+
/** Cancels the nudge — real model output arrived (or the session is resetting). */
|
|
637
|
+
cancelToolResultNudge() {
|
|
638
|
+
if (this.toolResultNudgeTimer) {
|
|
639
|
+
clearTimeout(this.toolResultNudgeTimer);
|
|
640
|
+
this.toolResultNudgeTimer = null;
|
|
641
|
+
}
|
|
642
|
+
}
|
|
643
|
+
/**
|
|
644
|
+
* Parses a JSON-stringified tool result into a structured value for the `result` slot,
|
|
645
|
+
* falling back to the raw string so a free-text result still round-trips.
|
|
646
|
+
*/
|
|
647
|
+
static ParseToolOutput(output) {
|
|
648
|
+
try {
|
|
649
|
+
return JSON.parse(output);
|
|
650
|
+
}
|
|
651
|
+
catch {
|
|
652
|
+
return output;
|
|
653
|
+
}
|
|
654
|
+
}
|
|
655
|
+
// ── Helpers ────────────────────────────────────────────────────────────────
|
|
656
|
+
/** JSON-serializes and sends one client frame (no-op once the socket is gone). */
|
|
657
|
+
sendFrame(frame) {
|
|
658
|
+
this.socket?.send(JSON.stringify(frame));
|
|
659
|
+
}
|
|
660
|
+
/** Resets the per-session response state machine (used on Disconnect). */
|
|
661
|
+
resetResponseState() {
|
|
662
|
+
this.responseActive = false;
|
|
663
|
+
this.activeResponseKind = 'normal';
|
|
664
|
+
this.queuedSends = [];
|
|
665
|
+
this.pendingToolCallIds.clear();
|
|
666
|
+
this.onInitiationMetadata = null;
|
|
667
|
+
}
|
|
668
|
+
/** Updates the client's own state view and emits the change to the host. */
|
|
669
|
+
setState(state) {
|
|
670
|
+
this.currentState = state;
|
|
671
|
+
this.emitStateChange(state);
|
|
672
|
+
}
|
|
673
|
+
};
|
|
674
|
+
ElevenLabsRealtimeClient = ElevenLabsRealtimeClient_1 = __decorate([
|
|
675
|
+
RegisterClass(BaseRealtimeClient, 'elevenlabs')
|
|
676
|
+
], ElevenLabsRealtimeClient);
|
|
677
|
+
export { ElevenLabsRealtimeClient };
|
|
678
|
+
/**
|
|
679
|
+
* Tree-shaking prevention: bundlers cannot see that {@link ElevenLabsRealtimeClient} is
|
|
680
|
+
* instantiated dynamically through the ClassFactory, so a consumer must call this no-op
|
|
681
|
+
* to create a static code path that keeps the `@RegisterClass` side effect alive.
|
|
682
|
+
*/
|
|
683
|
+
export function LoadElevenLabsRealtimeClient() {
|
|
684
|
+
// intentional no-op — the static import of this module is the point
|
|
685
|
+
}
|
|
686
|
+
//# sourceMappingURL=elevenLabsRealtimeClient.js.map
|