realtime-voice-agents 2.3.0 → 2.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/README.md +35 -12
  2. package/dist/{BaseRealtimeProvider-CWJ81HIt.d.mts → BaseRealtimeProvider-BJQO-hpu.d.mts} +44 -1
  3. package/dist/{BaseRealtimeProvider-BL75_HHh.d.cts → BaseRealtimeProvider-CO9N8jt-.d.cts} +44 -1
  4. package/dist/{BaseRealtimeProvider-C9C3s8jx.mjs → BaseRealtimeProvider-Cj6JWR2k.mjs} +1 -1
  5. package/dist/{BaseRealtimeProvider-C2mRn1V_.cjs → BaseRealtimeProvider-ZfFzijdr.cjs} +1 -1
  6. package/dist/{GeminiLiveProvider-Cc1dkxW6.d.cts → GeminiLiveProvider-C_LQCTX5.d.mts} +1 -1
  7. package/dist/{GeminiLiveProvider-TvY_cQQZ.d.mts → GeminiLiveProvider-DM-GwD0w.d.cts} +1 -1
  8. package/dist/{InMemorySessionStore-B5_rq61L.d.cts → InMemorySessionStore-BlsyPnIp.d.cts} +12 -6
  9. package/dist/{InMemorySessionStore-B5_rq61L.d.mts → InMemorySessionStore-BlsyPnIp.d.mts} +12 -6
  10. package/dist/{OpenAICompatibleProvider-D-2OOVBU.mjs → OpenAICompatibleProvider-Dqje-boq.mjs} +3 -3
  11. package/dist/{OpenAICompatibleProvider-Mp0Mefbh.cjs → OpenAICompatibleProvider-j2Nza2q2.cjs} +8 -2
  12. package/dist/{events-BylBSBW-.mjs → events-D-pvCH_F.mjs} +2 -1
  13. package/dist/{events-BxDTIKKq.cjs → events-DWX9535M.cjs} +2 -1
  14. package/dist/gemini.cjs +1 -1
  15. package/dist/gemini.d.cts +2 -2
  16. package/dist/gemini.d.mts +2 -2
  17. package/dist/gemini.mjs +1 -1
  18. package/dist/gpt-live.cjs +954 -0
  19. package/dist/gpt-live.d.cts +261 -0
  20. package/dist/gpt-live.d.mts +261 -0
  21. package/dist/gpt-live.mjs +936 -0
  22. package/dist/index.cjs +194 -25
  23. package/dist/index.d.cts +32 -3
  24. package/dist/index.d.mts +32 -3
  25. package/dist/index.mjs +194 -25
  26. package/dist/openai.cjs +2 -2
  27. package/dist/openai.d.cts +1 -1
  28. package/dist/openai.d.mts +1 -1
  29. package/dist/openai.mjs +2 -2
  30. package/dist/{rest-BYqiVOhe.mjs → rest-D-DWWVIj.mjs} +1 -1
  31. package/dist/{rest-BvUKut_k.cjs → rest-OR4VhEif.cjs} +1 -1
  32. package/dist/{session-config-CbifLlkV.mjs → session-config-C31WZESg.mjs} +1 -1
  33. package/dist/{session-config-CqJm2Kxz.cjs → session-config-CYjxBqpF.cjs} +1 -1
  34. package/dist/store.d.cts +2 -2
  35. package/dist/store.d.mts +2 -2
  36. package/dist/testing.cjs +376 -0
  37. package/dist/testing.d.cts +119 -2
  38. package/dist/testing.d.mts +119 -2
  39. package/dist/testing.mjs +374 -2
  40. package/dist/twilio.cjs +1 -1
  41. package/dist/twilio.mjs +1 -1
  42. package/dist/xai.cjs +3 -3
  43. package/dist/xai.d.cts +1 -1
  44. package/dist/xai.d.mts +1 -1
  45. package/dist/xai.mjs +3 -3
  46. package/package.json +14 -2
@@ -0,0 +1,954 @@
1
+ Object.defineProperty(exports, Symbol.toStringTag, { value: "Module" });
2
+ const require_rolldown_runtime = require("./rolldown-runtime-VH7oDXx4.cjs");
3
+ const require_BaseRealtimeProvider = require("./BaseRealtimeProvider-ZfFzijdr.cjs");
4
+ const require_mulaw = require("./mulaw-DLUObjdP.cjs");
5
+ const require_env = require("./env-DSnGaERV.cjs");
6
+ const require_OpenAICompatibleProvider = require("./OpenAICompatibleProvider-j2Nza2q2.cjs");
7
+ let ws = require("ws");
8
+ ws = require_rolldown_runtime.__toESM(ws, 1);
9
+ //#region src/providers/gpt-live/session-config.ts
10
+ /**
11
+ * `session.start` construction for GPT-Live (`/v1/live/sessions`).
12
+ *
13
+ * The session object is a STRICT schema: an unknown field rejects the whole
14
+ * start with `unknown_parameter` (field-verified Sept 2026), so every key
15
+ * here is a documented one and `providerOptions` / `extraSessionOptions`
16
+ * are deep-merged last as a deliberate escape hatch. Wire facts baked in:
17
+ * - one audio format for both directions, `{ type: 'audio/pcmu', rate: 8000 }`
18
+ * (Twilio's wire format — no transcoding)
19
+ * - instructions, voice, format and history are immutable after start
20
+ * - tools do not live on the voice model: they are `delegation.responses.tools`
21
+ * for the backend Responses model that reasons and calls them
22
+ * - history is `input`: ≤ 128 text messages / ≤ 8,192 rendered tokens
23
+ */
24
+ const GPT_LIVE_DEFAULT_BACKEND_MODEL = "gpt-5.6-terra";
25
+ const GPT_LIVE_AUDIO_FORMAT = {
26
+ type: "audio/pcmu",
27
+ rate: 8e3
28
+ };
29
+ const HISTORY_MAX_MESSAGES = 128;
30
+ /** ≈ 8,192 tokens at a conservative 2.5 chars/token (Hebrew and other dense scripts). */
31
+ const HISTORY_MAX_CHARS = 2e4;
32
+ function toBackendTool(tool) {
33
+ return {
34
+ type: "function",
35
+ name: tool.name,
36
+ description: tool.description ?? "",
37
+ parameters: tool.parameters
38
+ };
39
+ }
40
+ function buildDelegation(tools, options = {}) {
41
+ const responses = {
42
+ model: options.model ?? "gpt-5.6-terra",
43
+ tools: [...(tools ?? []).map(toBackendTool), ...options.extraTools ?? []],
44
+ tool_choice: options.toolChoice ?? "auto"
45
+ };
46
+ if (options.instructions !== void 0) responses.instructions = options.instructions;
47
+ if (options.parallelToolCalls !== void 0) responses.parallel_tool_calls = options.parallelToolCalls;
48
+ if (options.maxOutputTokens !== void 0) responses.max_output_tokens = options.maxOutputTokens;
49
+ if (options.reasoning !== void 0) responses.reasoning = options.reasoning;
50
+ if (options.serviceTier !== void 0) responses.service_tier = options.serviceTier;
51
+ if (options.text !== void 0) responses.text = options.text;
52
+ return {
53
+ type: "responses",
54
+ responses
55
+ };
56
+ }
57
+ /**
58
+ * History → `input` items, newest-first trimming to the API bounds. A leading
59
+ * `developer` entry (the engine's continuation note) is pinned so trimming
60
+ * never drops the context that explains the rest.
61
+ */
62
+ function buildHistoryItems(history, limits = {}) {
63
+ if (!history || history.length === 0) return {
64
+ items: [],
65
+ dropped: 0
66
+ };
67
+ const maxMessages = limits.maxMessages ?? HISTORY_MAX_MESSAGES;
68
+ const maxChars = limits.maxChars ?? HISTORY_MAX_CHARS;
69
+ const pinned = history[0].role === "developer" ? [history[0]] : [];
70
+ const rest = history.slice(pinned.length);
71
+ let chars = pinned.reduce((n, e) => n + e.text.length, 0);
72
+ const kept = [];
73
+ for (let i = rest.length - 1; i >= 0; i--) {
74
+ const entry = rest[i];
75
+ if (kept.length + pinned.length >= maxMessages) break;
76
+ if (chars + entry.text.length > maxChars) break;
77
+ chars += entry.text.length;
78
+ kept.unshift(entry);
79
+ }
80
+ return {
81
+ items: [...pinned, ...kept].map((entry) => ({
82
+ type: "message",
83
+ role: entry.role,
84
+ content: [{
85
+ type: entry.role === "assistant" ? "output_text" : "input_text",
86
+ text: entry.text
87
+ }]
88
+ })),
89
+ dropped: rest.length - kept.length
90
+ };
91
+ }
92
+ function buildSessionStart(init, config, eventId, model) {
93
+ const { items, dropped } = buildHistoryItems(init.history, {
94
+ maxMessages: config.historyMaxMessages,
95
+ maxChars: config.historyMaxChars
96
+ });
97
+ const voice = init.voice ?? config.defaultVoice;
98
+ let session = {
99
+ model,
100
+ instructions: init.instructions,
101
+ audio: {
102
+ format: { ...GPT_LIVE_AUDIO_FORMAT },
103
+ ...voice ? { output: { voice } } : {}
104
+ },
105
+ delegation: buildDelegation(init.tools, config.delegation),
106
+ ...items.length ? { input: items } : {},
107
+ ...config.store ? { store: true } : {}
108
+ };
109
+ session = require_BaseRealtimeProvider.deepMerge(session, init.providerOptions);
110
+ session = require_BaseRealtimeProvider.deepMerge(session, config.extraSessionOptions);
111
+ return {
112
+ frame: {
113
+ type: "session.start",
114
+ event_id: eventId,
115
+ session
116
+ },
117
+ droppedHistory: dropped
118
+ };
119
+ }
120
+ //#endregion
121
+ //#region src/providers/gpt-live/speech-gate.ts
122
+ /**
123
+ * Speech gate for full-duplex output streams.
124
+ *
125
+ * GPT-Live streams assistant audio continuously and in real time — silence
126
+ * included (≈100 ms μ-law deltas, digital 0xFF while the model is quiet) —
127
+ * and the wire has no response boundaries: no response.created, no
128
+ * output_audio.done. The engine's playback tracking, tool timing and hangup
129
+ * watchdog all key off `responseStarted` / `responseDone`, so this gate
130
+ * synthesizes utterances from the audio itself: one opens on the first loud
131
+ * 20 ms sub-frame and closes after `quietMs` of quiet AUDIO. Quiet is
132
+ * measured on the stream's own clock (bytes received), never wall-clock, so
133
+ * a stalled socket cannot fake a finished utterance and tests can script
134
+ * silence instantly. Field numbers (Sept 2026, 8 kHz μ-law): quiet frames
135
+ * p99 ≈ −50 dBFS, speech p05 ≈ −44 dBFS; LiveKit and Pipecat both close at
136
+ * 0.8 s.
137
+ */
138
+ const DEFAULT_GATE_THRESHOLD_RMS = .004;
139
+ const DEFAULT_GATE_QUIET_MS = 800;
140
+ const SUBFRAME_BYTES = 160;
141
+ const BYTES_PER_MS = 8;
142
+ var SpeechGate = class {
143
+ threshold;
144
+ quietLimitMs;
145
+ open = false;
146
+ quietMs = 0;
147
+ utteranceMs = 0;
148
+ constructor(options = {}) {
149
+ this.threshold = options.thresholdRms ?? .004;
150
+ this.quietLimitMs = options.quietMs ?? 800;
151
+ }
152
+ get isOpen() {
153
+ return this.open;
154
+ }
155
+ /** Feed one output delta; returns the transitions it caused, in order. */
156
+ feed(bytes) {
157
+ const events = [];
158
+ for (let offset = 0; offset < bytes.length; offset += SUBFRAME_BYTES) {
159
+ const frame = bytes.subarray(offset, Math.min(offset + SUBFRAME_BYTES, bytes.length));
160
+ const frameMs = frame.length / BYTES_PER_MS;
161
+ if (isSpeech(frame, this.threshold)) {
162
+ if (!this.open) {
163
+ this.open = true;
164
+ this.utteranceMs = 0;
165
+ events.push({ type: "open" });
166
+ }
167
+ this.quietMs = 0;
168
+ this.utteranceMs += frameMs;
169
+ } else if (this.open) {
170
+ this.quietMs += frameMs;
171
+ this.utteranceMs += frameMs;
172
+ if (this.quietMs >= this.quietLimitMs) {
173
+ events.push({
174
+ type: "close",
175
+ utteranceMs: this.utteranceMs - this.quietMs
176
+ });
177
+ this.open = false;
178
+ this.quietMs = 0;
179
+ this.utteranceMs = 0;
180
+ }
181
+ }
182
+ }
183
+ return events;
184
+ }
185
+ /** Force-close an open utterance (stream stalled, session closing). */
186
+ close() {
187
+ if (!this.open) return null;
188
+ const event = {
189
+ type: "close",
190
+ utteranceMs: this.utteranceMs - this.quietMs
191
+ };
192
+ this.open = false;
193
+ this.quietMs = 0;
194
+ this.utteranceMs = 0;
195
+ return event;
196
+ }
197
+ };
198
+ /** μ-law digital zero is 0xFF (some encoders emit 0x7F for negative zero). */
199
+ function isDigitalSilence(bytes) {
200
+ for (let i = 0; i < bytes.length; i++) {
201
+ const b = bytes[i];
202
+ if (b !== 255 && b !== 127) return false;
203
+ }
204
+ return true;
205
+ }
206
+ function isSpeech(frame, threshold) {
207
+ if (frame.length === 0 || isDigitalSilence(frame)) return false;
208
+ const pcm = require_mulaw.mulawToPcm16(frame);
209
+ let acc = 0;
210
+ for (let i = 0; i < pcm.length; i++) {
211
+ const v = pcm[i] / 32768;
212
+ acc += v * v;
213
+ }
214
+ return Math.sqrt(acc / pcm.length) >= threshold;
215
+ }
216
+ //#endregion
217
+ //#region src/providers/gpt-live/transcript-grouper.ts
218
+ var TranscriptGrouper = class {
219
+ options;
220
+ current = null;
221
+ idleTimer = null;
222
+ constructor(options) {
223
+ this.options = options;
224
+ }
225
+ push(delta, startMs, endMs) {
226
+ if (this.current && startMs - this.current.endMs > this.options.gapMs) this.flush();
227
+ if (!this.current) this.current = {
228
+ text: "",
229
+ startMs,
230
+ endMs,
231
+ tag: this.options.tagFor?.()
232
+ };
233
+ this.current.text += delta;
234
+ this.current.endMs = Math.max(this.current.endMs, endMs);
235
+ this.armIdle();
236
+ }
237
+ /** Close the open turn now (session ending, utterance boundary reached). */
238
+ flush() {
239
+ this.clearIdle();
240
+ const turn = this.current;
241
+ this.current = null;
242
+ if (!turn) return;
243
+ const text = turn.text.trim();
244
+ if (text.length === 0) return;
245
+ this.options.onTurn({
246
+ ...turn,
247
+ text
248
+ });
249
+ }
250
+ dispose() {
251
+ this.clearIdle();
252
+ this.current = null;
253
+ }
254
+ armIdle() {
255
+ this.clearIdle();
256
+ this.idleTimer = setTimeout(() => {
257
+ this.idleTimer = null;
258
+ this.flush();
259
+ }, this.options.idleMs);
260
+ this.idleTimer.unref?.();
261
+ }
262
+ clearIdle() {
263
+ if (this.idleTimer) {
264
+ clearTimeout(this.idleTimer);
265
+ this.idleTimer = null;
266
+ }
267
+ }
268
+ };
269
+ //#endregion
270
+ //#region src/providers/gpt-live/GptLiveProvider.ts
271
+ /**
272
+ * GPT-Live provider (`/v1/live/sessions`, raw WebSocket) — OpenAI's
273
+ * full-duplex speech-to-speech model. A different API from Realtime, not a
274
+ * new Realtime model: its own handshake (`session.start` → `session.started`),
275
+ * its own events, and a different division of labour that this provider
276
+ * translates into the engine's normalized contract:
277
+ *
278
+ * - **Turn-taking is the model's.** It listens while it speaks and stops on
279
+ * its own when the caller talks. There is no VAD config, no cancel, no
280
+ * truncate, no speech_started/stopped; `capabilities.turnTaking: 'model'`
281
+ * tells the engine to keep its hands off (see parity tests).
282
+ * - **Output is a continuous real-time stream, silence included** (100 ms
283
+ * μ-law deltas, digital 0xFF when quiet). Response boundaries are
284
+ * synthesized by a speech gate on the audio itself (`speech-gate.ts`);
285
+ * silence between utterances is not forwarded.
286
+ * - **The backend does the thinking.** Tools are declared on a Responses
287
+ * model (`delegation.responses`), calls arrive wrapped in `response.event`
288
+ * envelopes, results go back as `response.item.create` + one
289
+ * `response.create`. The voice keeps talking meanwhile
290
+ * (`capabilities.decoupledBackend`).
291
+ * - **Text reaches the model only through appends** (≤ 500 tokens each):
292
+ * instructions (policy), thinking (quiet context), commentary (say this).
293
+ * `createResponse({ instructions })` maps to commentary — field-tested as
294
+ * the only append that reliably produces speech on demand (Sept 2026).
295
+ * - **Instructions, voice and history are immutable after start** — handoffs
296
+ * reconnect and seed the attributed transcript via `session.input`
297
+ * (`capabilities.startupHistory`).
298
+ * - **Billing is per second of session**, reported as cumulative
299
+ * `session.usage.updated` ticks; `close()` sends `session.close` and waits
300
+ * for `session.closed` so the final usage is confirmed.
301
+ */
302
+ const GPT_LIVE_DEFAULT_BASE_URL = "wss://api.openai.com/v1/live/sessions";
303
+ const NON_RETRIABLE_CLOSE_CODES = /* @__PURE__ */ new Set([
304
+ 1002,
305
+ 1003,
306
+ 1007,
307
+ 1008
308
+ ]);
309
+ /** Appends are capped at 500 tokens; split conservatively (dense scripts run ~2.5 chars/token). */
310
+ const APPEND_MAX_CHARS = 1200;
311
+ const TRANSCRIPT_IDLE_EXTRA_MS = 300;
312
+ const GATE_STALL_EXTRA_MS = 500;
313
+ const ACK_TYPES = {
314
+ "session.instructions.append": "session.instructions.appended",
315
+ "session.thinking.append": "session.thinking.appended",
316
+ "session.commentary.append": "session.commentary.appended",
317
+ "session.update": "session.updated",
318
+ "session.input_audio.mute": "session.input_audio.muted",
319
+ "session.input_audio.unmute": "session.input_audio.unmuted"
320
+ };
321
+ let eventSeq = 0;
322
+ const nextEventId = (prefix) => `${prefix}_${++eventSeq}`;
323
+ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimeProvider {
324
+ name;
325
+ capabilities = {
326
+ truncate: false,
327
+ sessionUpdate: false,
328
+ voiceChangeMidSession: false,
329
+ transcodeRequired: false,
330
+ resumption: false,
331
+ agentTranscriptDeltas: true,
332
+ vadInterruptControl: false,
333
+ turnTaking: "model",
334
+ startupHistory: true,
335
+ decoupledBackend: true
336
+ };
337
+ config;
338
+ logger;
339
+ ws = null;
340
+ ready = false;
341
+ intentionalClose = false;
342
+ sessionInit = null;
343
+ /** Server-assigned session id (sideband attach, forking, recording download). */
344
+ sessionId = null;
345
+ /** Unix seconds at which the server expires the session (120 min at launch). */
346
+ expiresAt = null;
347
+ gate;
348
+ gateStallTimer = null;
349
+ utteranceCounter = 0;
350
+ currentUtteranceId = null;
351
+ lastUtteranceId = null;
352
+ inputTranscript;
353
+ outputTranscript;
354
+ acks = /* @__PURE__ */ new Map();
355
+ seenCalls = /* @__PURE__ */ new Set();
356
+ sessionClosed = null;
357
+ closeListeners = [];
358
+ constructor(config, logger = require_BaseRealtimeProvider.noopLogger) {
359
+ super();
360
+ this.config = config;
361
+ this.logger = logger;
362
+ this.name = config.providerName ?? "gpt-live";
363
+ this.gate = new SpeechGate(config.speechGate);
364
+ const gapMs = config.transcriptGapMs ?? 800;
365
+ this.inputTranscript = new TranscriptGrouper({
366
+ gapMs,
367
+ idleMs: gapMs + TRANSCRIPT_IDLE_EXTRA_MS,
368
+ onTurn: (turn) => this.emit("userTranscript", { text: turn.text })
369
+ });
370
+ this.outputTranscript = new TranscriptGrouper({
371
+ gapMs,
372
+ idleMs: gapMs + TRANSCRIPT_IDLE_EXTRA_MS,
373
+ tagFor: () => this.currentUtteranceId ?? this.lastUtteranceId ?? void 0,
374
+ onTurn: (turn) => this.emit("agentTranscript", {
375
+ responseId: turn.tag ?? "",
376
+ text: turn.text
377
+ })
378
+ });
379
+ }
380
+ get isConnected() {
381
+ return this.ready && this.ws?.readyState === ws.default.OPEN;
382
+ }
383
+ async connect(init) {
384
+ if (this.ws) await this.close();
385
+ this.sessionInit = init;
386
+ this.intentionalClose = false;
387
+ this.ready = false;
388
+ this.sessionClosed = null;
389
+ this.sessionId = null;
390
+ this.expiresAt = null;
391
+ this.resetUtteranceState();
392
+ if (init.vad !== void 0 || init.transcription !== void 0 || init.temperature !== void 0) this.logger.debug("gpt-live: vad/transcription/temperature are model-owned and ignored");
393
+ const startId = nextEventId("start");
394
+ const { frame, droppedHistory } = buildSessionStart(init, this.sessionConfig(), startId, this.config.model);
395
+ if (droppedHistory > 0) this.logger.warn("gpt-live: startup history trimmed to the API bounds", { dropped: droppedHistory });
396
+ const ws$1 = new ws.default(this.config.baseUrl ?? "wss://api.openai.com/v1/live/sessions", {
397
+ headers: {
398
+ Authorization: `Bearer ${this.config.apiKey}`,
399
+ ...this.config.headers
400
+ },
401
+ perMessageDeflate: false
402
+ });
403
+ this.ws = ws$1;
404
+ await new Promise((resolve, reject) => {
405
+ const timeoutMs = this.config.connectTimeoutMs ?? 15e3;
406
+ const timer = setTimeout(() => fail(/* @__PURE__ */ new Error(`${this.name} session not started within ${timeoutMs}ms`)), timeoutMs);
407
+ timer.unref?.();
408
+ let settled = false;
409
+ const fail = (error) => {
410
+ if (settled) return;
411
+ settled = true;
412
+ clearTimeout(timer);
413
+ try {
414
+ ws$1.close();
415
+ } catch {}
416
+ reject(error);
417
+ };
418
+ ws$1.on("unexpected-response", (_request, response) => {
419
+ let body = "";
420
+ response.setEncoding("utf8");
421
+ response.on("data", (chunk) => {
422
+ if (body.length < 512) body += chunk;
423
+ });
424
+ response.on("end", () => {
425
+ const detail = body.trim().slice(0, 500);
426
+ this.logger.error("provider rejected the WebSocket upgrade", {
427
+ status: response.statusCode,
428
+ body: detail
429
+ });
430
+ fail(/* @__PURE__ */ new Error(`${this.name} rejected the WebSocket upgrade: HTTP ${response.statusCode}${detail ? ` — ${detail}` : ""}`));
431
+ });
432
+ });
433
+ ws$1.on("open", () => {
434
+ this.emit("open");
435
+ this.send(frame);
436
+ });
437
+ ws$1.on("message", (raw) => {
438
+ const event = this.parseEvent(raw);
439
+ if (!event) return;
440
+ if (!settled) {
441
+ if (event.type === "session.started") {
442
+ settled = true;
443
+ clearTimeout(timer);
444
+ this.sessionId = event.session?.id ?? null;
445
+ this.expiresAt = typeof event.session?.expires_at === "number" ? event.session.expires_at : null;
446
+ this.ready = true;
447
+ this.logger.info("gpt-live session started", {
448
+ sessionId: this.sessionId,
449
+ expiresAt: this.expiresAt ? (/* @__PURE__ */ new Date(this.expiresAt * 1e3)).toISOString() : void 0
450
+ });
451
+ resolve();
452
+ return;
453
+ }
454
+ if (event.type === "error") {
455
+ fail(/* @__PURE__ */ new Error(`${this.name} session.start rejected: ${JSON.stringify(event.error ?? event)}`));
456
+ return;
457
+ }
458
+ return;
459
+ }
460
+ this.handleEvent(event);
461
+ });
462
+ ws$1.on("error", (error) => {
463
+ const err = error instanceof Error ? error : new Error(String(error));
464
+ if (!settled) fail(err);
465
+ else this.emit("error", err);
466
+ });
467
+ ws$1.on("close", (code, reasonBuf) => {
468
+ const reason = reasonBuf?.toString();
469
+ this.ready = false;
470
+ this.failPendingAcks();
471
+ this.finishUtteranceOnClose();
472
+ const listeners = this.closeListeners;
473
+ this.closeListeners = [];
474
+ for (const listener of listeners) listener();
475
+ if (!settled) {
476
+ fail(/* @__PURE__ */ new Error(`${this.name} socket closed during setup (${code} ${reason ?? ""})`));
477
+ return;
478
+ }
479
+ this.emit("close", {
480
+ code,
481
+ reason,
482
+ retriable: this.isRetriableClose(code)
483
+ });
484
+ });
485
+ });
486
+ }
487
+ async close() {
488
+ this.intentionalClose = true;
489
+ this.ready = false;
490
+ const ws$2 = this.ws;
491
+ this.ws = null;
492
+ this.inputTranscript.dispose();
493
+ this.outputTranscript.dispose();
494
+ this.clearGateStall();
495
+ if (!ws$2 || ws$2.readyState === ws.default.CLOSED) return;
496
+ await new Promise((resolve) => {
497
+ const timer = setTimeout(() => finish(), this.config.closeTimeoutMs ?? 3e3);
498
+ timer.unref?.();
499
+ let done = false;
500
+ const finish = () => {
501
+ if (done) return;
502
+ done = true;
503
+ clearTimeout(timer);
504
+ try {
505
+ if (ws$2.readyState === ws.default.OPEN || ws$2.readyState === ws.default.CONNECTING) ws$2.close(1e3);
506
+ } catch {}
507
+ setTimeout(() => {
508
+ try {
509
+ ws$2.terminate();
510
+ } catch {}
511
+ }, 1e3).unref?.();
512
+ resolve();
513
+ };
514
+ this.closeListeners.push(finish);
515
+ if (ws$2.readyState === ws.default.OPEN) try {
516
+ ws$2.send(JSON.stringify({
517
+ type: "session.close",
518
+ event_id: nextEventId("close")
519
+ }));
520
+ } catch {
521
+ finish();
522
+ }
523
+ else finish();
524
+ });
525
+ }
526
+ sendAudio(base64Mulaw) {
527
+ if (!this.isConnected) return;
528
+ this.send({
529
+ type: "session.input_audio.append",
530
+ audio: base64Mulaw
531
+ });
532
+ }
533
+ /**
534
+ * Text has three doors into a GPT-Live session, by intent rather than by
535
+ * conversation role: `system` → instructions (behaviour), `user` → thinking
536
+ * (quiet context: keypad entries, deferred results — the model decides how
537
+ * to react), `assistant` → thinking as well, worded as something already
538
+ * said (commentary would make the model say it again).
539
+ */
540
+ sendText(text, options = {}) {
541
+ const role = options.role ?? "user";
542
+ if (role === "system") this.append("session.instructions.append", text);
543
+ else if (role === "assistant") this.append("session.thinking.append", `You already said this to the caller earlier: "${text}"`);
544
+ else this.append("session.thinking.append", text);
545
+ }
546
+ sendToolResult(callId, output, options = {}) {
547
+ this.send({
548
+ type: "response.item.create",
549
+ event_id: nextEventId("tool"),
550
+ item: {
551
+ type: "function_call_output",
552
+ call_id: callId,
553
+ output: typeof output === "string" ? output : JSON.stringify(output ?? null)
554
+ }
555
+ });
556
+ if (options.triggerResponse !== false) this.send({
557
+ type: "response.create",
558
+ event_id: nextEventId("continue")
559
+ });
560
+ }
561
+ /**
562
+ * "Speak now" has no direct verb on this API. Commentary — information the
563
+ * model should say aloud, paraphrasing allowed — is the append that
564
+ * produced speech on demand in every field test, greeting and goodbye
565
+ * alike; an instruction-shaped text is followed rather than read out.
566
+ */
567
+ createResponse(options = {}) {
568
+ if (options.instructions) this.append("session.commentary.append", options.instructions);
569
+ else this.append("session.instructions.append", "Respond to the caller now, without waiting for them to speak.");
570
+ }
571
+ /**
572
+ * Only the backend delegation can change mid-session (`session.update` is
573
+ * sparse and limited to `delegation.responses`): tool lists land there;
574
+ * instructions / voice / vad are immutable and are reported, not applied.
575
+ */
576
+ async updateSession(patch, options = {}) {
577
+ if (!this.sessionInit) throw new Error("updateSession before connect");
578
+ const ignored = Object.keys(patch).filter((key) => key !== "tools" && key !== "providerOptions");
579
+ if (ignored.length > 0) this.logger.warn("gpt-live: session fields are immutable after start — ignored", { fields: ignored });
580
+ this.sessionInit = {
581
+ ...this.sessionInit,
582
+ ...patch
583
+ };
584
+ let responses = {};
585
+ if (patch.tools) responses.tools = [...patch.tools.map(toBackendTool), ...this.config.delegation?.extraTools ?? []];
586
+ const nativeResponses = patch.providerOptions?.delegation?.responses;
587
+ if (nativeResponses && typeof nativeResponses === "object") responses = require_BaseRealtimeProvider.deepMerge(responses, nativeResponses);
588
+ if (Object.keys(responses).length === 0) return false;
589
+ if (!this.isConnected) return false;
590
+ const eventId = nextEventId("update");
591
+ const acked = this.awaitAck(eventId, ACK_TYPES["session.update"]);
592
+ this.send({
593
+ type: "session.update",
594
+ event_id: eventId,
595
+ session: { delegation: {
596
+ type: "responses",
597
+ responses
598
+ } }
599
+ });
600
+ return options.awaitAck ? acked : true;
601
+ }
602
+ handleEvent(event) {
603
+ switch (event.type) {
604
+ case "session.output_audio.delta":
605
+ if (typeof event.delta !== "string" || event.delta.length === 0) break;
606
+ this.handleOutputAudio(event.delta);
607
+ break;
608
+ case "session.input_transcript.delta":
609
+ if (typeof event.delta === "string") this.inputTranscript.push(event.delta, numberOr(event.start_ms, 0), numberOr(event.end_ms, 0));
610
+ break;
611
+ case "session.output_transcript.delta":
612
+ if (typeof event.delta === "string") {
613
+ this.emit("agentTranscriptDelta", {
614
+ responseId: this.currentUtteranceId ?? this.lastUtteranceId ?? "",
615
+ delta: event.delta
616
+ });
617
+ this.outputTranscript.push(event.delta, numberOr(event.start_ms, 0), numberOr(event.end_ms, 0));
618
+ }
619
+ break;
620
+ case "session.delegation.created":
621
+ this.logger.debug("gpt-live delegation created", {
622
+ id: event.delegation?.id,
623
+ target: event.delegation?.target,
624
+ responseId: event.delegation?.response_id
625
+ });
626
+ break;
627
+ case "response.event":
628
+ this.handleBackendEvent(event.event ?? {}, event.delegation_id);
629
+ break;
630
+ case "session.usage.updated": {
631
+ const seconds = event.usage?.seconds;
632
+ if (typeof seconds === "number") this.emit("usage", {
633
+ inputTokens: 0,
634
+ outputTokens: 0,
635
+ totalTokens: 0,
636
+ audioSeconds: seconds,
637
+ raw: event
638
+ });
639
+ break;
640
+ }
641
+ case "session.instructions.appended":
642
+ case "session.thinking.appended":
643
+ case "session.commentary.appended":
644
+ case "session.updated":
645
+ case "session.input_audio.muted":
646
+ case "session.input_audio.unmuted":
647
+ this.resolveAck(event.client_event_id, true);
648
+ break;
649
+ case "session.closed": {
650
+ const seconds = event.usage?.seconds;
651
+ this.sessionClosed = {
652
+ reason: event.reason,
653
+ usageSeconds: typeof seconds === "number" ? seconds : void 0
654
+ };
655
+ this.logger.info("gpt-live session closed", {
656
+ reason: event.reason,
657
+ usageSeconds: seconds
658
+ });
659
+ if (typeof seconds === "number") this.emit("usage", {
660
+ inputTokens: 0,
661
+ outputTokens: 0,
662
+ totalTokens: 0,
663
+ audioSeconds: seconds,
664
+ raw: event
665
+ });
666
+ break;
667
+ }
668
+ case "error":
669
+ if (event.error?.client_event_id) this.resolveAck(event.error.client_event_id, false);
670
+ this.logger.warn("provider error event", { error: event.error });
671
+ this.emit("error", /* @__PURE__ */ new Error(`${this.name} error: ${JSON.stringify(event.error ?? event)}`));
672
+ break;
673
+ case "info": this.logger.debug("gpt-live info", {
674
+ code: event.code,
675
+ message: event.message
676
+ });
677
+ }
678
+ }
679
+ handleOutputAudio(delta) {
680
+ const bytes = Buffer.from(delta, "base64");
681
+ const wasOpen = this.gate.isOpen;
682
+ const events = this.gate.feed(bytes);
683
+ let forwardId = wasOpen ? this.currentUtteranceId : null;
684
+ let close = null;
685
+ for (const gateEvent of events) if (gateEvent.type === "open") {
686
+ this.beginUtterance();
687
+ forwardId = this.currentUtteranceId;
688
+ } else close = gateEvent;
689
+ if (forwardId) this.emit("audio", {
690
+ base64Mulaw: delta,
691
+ responseId: forwardId
692
+ });
693
+ if (close) this.endUtterance();
694
+ else if (this.gate.isOpen) this.armGateStall();
695
+ }
696
+ handleBackendEvent(inner, delegationId) {
697
+ const type = inner.type ?? "";
698
+ if (type === "response.output_item.done" && inner.item?.type === "function_call") {
699
+ const item = inner.item;
700
+ const callId = item.call_id ?? item.id ?? `call_${Date.now()}`;
701
+ if (this.seenCalls.has(callId)) return;
702
+ this.seenCalls.add(callId);
703
+ if (this.seenCalls.size > 1e3) this.seenCalls.delete(this.seenCalls.values().next().value);
704
+ this.emit("toolCall", {
705
+ id: callId,
706
+ name: item.name,
707
+ argumentsJson: typeof item.arguments === "string" ? item.arguments : "{}",
708
+ responseId: delegationId,
709
+ itemId: item.id
710
+ });
711
+ return;
712
+ }
713
+ if (type === "response.completed" || type === "response.done") {
714
+ const usage = require_OpenAICompatibleProvider.normalizeUsage(inner.response?.usage);
715
+ if (usage) this.emit("usage", usage);
716
+ return;
717
+ }
718
+ if (type === "response.failed" || type === "response.incomplete") this.logger.warn("gpt-live backend response did not complete", {
719
+ type,
720
+ delegationId,
721
+ error: inner.response?.error
722
+ });
723
+ }
724
+ beginUtterance() {
725
+ this.currentUtteranceId = `live_utt_${++this.utteranceCounter}`;
726
+ this.lastUtteranceId = this.currentUtteranceId;
727
+ this.emit("responseStarted", { responseId: this.currentUtteranceId });
728
+ }
729
+ endUtterance() {
730
+ this.clearGateStall();
731
+ const id = this.currentUtteranceId;
732
+ this.currentUtteranceId = null;
733
+ if (id) this.emit("responseDone", { responseId: id });
734
+ }
735
+ /** The stream is real-time: no delta for a quiet window means it stalled — close the utterance. */
736
+ armGateStall() {
737
+ this.clearGateStall();
738
+ const quietMs = this.config.speechGate?.quietMs ?? 800;
739
+ this.gateStallTimer = setTimeout(() => {
740
+ this.gateStallTimer = null;
741
+ if (this.gate.close()) this.endUtterance();
742
+ }, quietMs + GATE_STALL_EXTRA_MS);
743
+ this.gateStallTimer.unref?.();
744
+ }
745
+ clearGateStall() {
746
+ if (this.gateStallTimer) {
747
+ clearTimeout(this.gateStallTimer);
748
+ this.gateStallTimer = null;
749
+ }
750
+ }
751
+ finishUtteranceOnClose() {
752
+ this.clearGateStall();
753
+ if (this.gate.close()) this.endUtterance();
754
+ this.inputTranscript.flush();
755
+ this.outputTranscript.flush();
756
+ }
757
+ resetUtteranceState() {
758
+ this.clearGateStall();
759
+ this.gate.close();
760
+ this.currentUtteranceId = null;
761
+ this.lastUtteranceId = null;
762
+ this.seenCalls.clear();
763
+ }
764
+ /** Send `content` as one or more appends (500-token cap); resolves with the last chunk's ack. */
765
+ async append(type, content, delegationId = null) {
766
+ if (!this.isConnected) return false;
767
+ const chunks = splitForAppend(content);
768
+ if (chunks.length > 1) this.logger.warn("gpt-live: append split to respect the 500-token cap", {
769
+ type,
770
+ chunks: chunks.length
771
+ });
772
+ let acked = false;
773
+ for (const chunk of chunks) {
774
+ const eventId = nextEventId("append");
775
+ const ack = this.awaitAck(eventId, ACK_TYPES[type]);
776
+ this.send({
777
+ type,
778
+ event_id: eventId,
779
+ delegation_id: delegationId,
780
+ content: chunk
781
+ });
782
+ acked = await ack;
783
+ if (!acked) break;
784
+ }
785
+ return acked;
786
+ }
787
+ awaitAck(eventId, ackType) {
788
+ return new Promise((resolve) => {
789
+ const timer = setTimeout(() => {
790
+ if (this.acks.delete(eventId)) {
791
+ this.logger.warn("gpt-live: no acknowledgement for client event", {
792
+ eventId,
793
+ ackType
794
+ });
795
+ resolve(false);
796
+ }
797
+ }, this.config.ackTimeoutMs ?? 5e3);
798
+ timer.unref?.();
799
+ this.acks.set(eventId, {
800
+ ackType,
801
+ resolve,
802
+ timer
803
+ });
804
+ });
805
+ }
806
+ resolveAck(clientEventId, acked) {
807
+ if (typeof clientEventId !== "string") return;
808
+ const waiter = this.acks.get(clientEventId);
809
+ if (!waiter) return;
810
+ this.acks.delete(clientEventId);
811
+ clearTimeout(waiter.timer);
812
+ waiter.resolve(acked);
813
+ }
814
+ failPendingAcks() {
815
+ for (const [id, waiter] of this.acks) {
816
+ clearTimeout(waiter.timer);
817
+ waiter.resolve(false);
818
+ this.acks.delete(id);
819
+ }
820
+ }
821
+ sessionConfig() {
822
+ return {
823
+ defaultVoice: this.config.voice,
824
+ delegation: this.config.delegation,
825
+ store: this.config.store,
826
+ extraSessionOptions: this.config.extraSessionOptions,
827
+ historyMaxMessages: this.config.historyMaxMessages,
828
+ historyMaxChars: this.config.historyMaxChars
829
+ };
830
+ }
831
+ isRetriableClose(code) {
832
+ if (this.intentionalClose) return false;
833
+ if (NON_RETRIABLE_CLOSE_CODES.has(code)) return false;
834
+ const reason = this.sessionClosed?.reason;
835
+ if (reason && reason !== "expired" && reason !== "connection_lost") return false;
836
+ return true;
837
+ }
838
+ parseEvent(raw) {
839
+ try {
840
+ return JSON.parse(raw.toString());
841
+ } catch {
842
+ this.logger.warn("unparseable provider frame");
843
+ return null;
844
+ }
845
+ }
846
+ send(payload) {
847
+ if (this.ws?.readyState !== ws.default.OPEN) return;
848
+ try {
849
+ this.ws.send(JSON.stringify(payload));
850
+ } catch (error) {
851
+ this.emit("error", error instanceof Error ? error : new Error(String(error)));
852
+ }
853
+ }
854
+ };
855
+ function numberOr(value, fallback) {
856
+ return typeof value === "number" ? value : fallback;
857
+ }
858
+ /** Split on whitespace so no chunk exceeds the per-append cap. */
859
+ function splitForAppend(content, maxChars = APPEND_MAX_CHARS) {
860
+ const text = content.trim();
861
+ if (text.length <= maxChars) return [text];
862
+ const chunks = [];
863
+ let rest = text;
864
+ while (rest.length > maxChars) {
865
+ let cut = rest.lastIndexOf(" ", maxChars);
866
+ if (cut < maxChars / 2) cut = maxChars;
867
+ chunks.push(rest.slice(0, cut).trim());
868
+ rest = rest.slice(cut).trim();
869
+ }
870
+ if (rest.length) chunks.push(rest);
871
+ return chunks;
872
+ }
873
+ //#endregion
874
+ //#region src/gpt-live.ts
875
+ /** OpenAI GPT-Live provider (full-duplex Live API) — `realtime-voice-agents/gpt-live`. */
876
+ /** Env vars checked (in order) when no explicit apiKey is passed — the same project key as Realtime. */
877
+ const GPT_LIVE_KEY_ENV_VARS = [
878
+ "OPENAI_API_KEY",
879
+ "OPENAI_KEY",
880
+ "OPEN_AI_API_KEY"
881
+ ];
882
+ const GPT_LIVE_DEFAULT_MODEL = "gpt-live-1";
883
+ const GPT_LIVE_DEFAULT_VOICE = "marin";
884
+ /** Built-in voices accepted by the Live API (the SDK's enum, Sept 2026). */
885
+ const GPT_LIVE_VOICES = [
886
+ "marin",
887
+ "cedar",
888
+ "alloy",
889
+ "ash",
890
+ "ballad",
891
+ "beacon",
892
+ "bossa",
893
+ "cinder",
894
+ "coral",
895
+ "delta",
896
+ "echo",
897
+ "gleam",
898
+ "meridian",
899
+ "quartz",
900
+ "ripple",
901
+ "sage",
902
+ "shimmer",
903
+ "stone",
904
+ "tempo",
905
+ "verse",
906
+ "vesper",
907
+ "willow"
908
+ ];
909
+ /**
910
+ * Create a GPT-Live provider factory for the bridge (`provider: gptLive({...})`).
911
+ *
912
+ * Credentials are resolved per call, when the factory runs (see
913
+ * `openaiRealtime` — a missing key fails that call's connect so a fallback
914
+ * chain can absorb it instead of crashing config construction).
915
+ */
916
+ function gptLive(options = {}) {
917
+ const config = {
918
+ model: options.model ?? "gpt-live-1",
919
+ voice: options.voice ?? "marin",
920
+ delegation: options.delegation,
921
+ baseUrl: options.baseUrl,
922
+ headers: options.headers,
923
+ store: options.store,
924
+ extraSessionOptions: options.sessionOptions,
925
+ connectTimeoutMs: options.connectTimeoutMs,
926
+ speechGate: options.speechGate,
927
+ transcriptGapMs: options.transcriptGapMs
928
+ };
929
+ return ({ logger }) => {
930
+ const apiKey = require_env.resolveApiKey(options.apiKey, GPT_LIVE_KEY_ENV_VARS);
931
+ if (!apiKey) throw new Error(`gptLive: apiKey missing (pass apiKey or set one of ${GPT_LIVE_KEY_ENV_VARS.join("/")})`);
932
+ return new GptLiveProvider({
933
+ ...config,
934
+ apiKey
935
+ }, logger);
936
+ };
937
+ }
938
+ //#endregion
939
+ exports.DEFAULT_GATE_QUIET_MS = DEFAULT_GATE_QUIET_MS;
940
+ exports.DEFAULT_GATE_THRESHOLD_RMS = DEFAULT_GATE_THRESHOLD_RMS;
941
+ exports.GPT_LIVE_AUDIO_FORMAT = GPT_LIVE_AUDIO_FORMAT;
942
+ exports.GPT_LIVE_DEFAULT_BACKEND_MODEL = GPT_LIVE_DEFAULT_BACKEND_MODEL;
943
+ exports.GPT_LIVE_DEFAULT_BASE_URL = GPT_LIVE_DEFAULT_BASE_URL;
944
+ exports.GPT_LIVE_DEFAULT_MODEL = GPT_LIVE_DEFAULT_MODEL;
945
+ exports.GPT_LIVE_DEFAULT_VOICE = GPT_LIVE_DEFAULT_VOICE;
946
+ exports.GPT_LIVE_KEY_ENV_VARS = GPT_LIVE_KEY_ENV_VARS;
947
+ exports.GPT_LIVE_VOICES = GPT_LIVE_VOICES;
948
+ exports.GptLiveProvider = GptLiveProvider;
949
+ exports.SpeechGate = SpeechGate;
950
+ exports.buildDelegation = buildDelegation;
951
+ exports.buildHistoryItems = buildHistoryItems;
952
+ exports.buildSessionStart = buildSessionStart;
953
+ exports.gptLive = gptLive;
954
+ exports.splitForAppend = splitForAppend;