realtime-voice-agents 2.3.0 → 2.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +35 -12
- package/dist/{BaseRealtimeProvider-CWJ81HIt.d.mts → BaseRealtimeProvider-BJQO-hpu.d.mts} +44 -1
- package/dist/{BaseRealtimeProvider-BL75_HHh.d.cts → BaseRealtimeProvider-CO9N8jt-.d.cts} +44 -1
- package/dist/{BaseRealtimeProvider-C9C3s8jx.mjs → BaseRealtimeProvider-Cj6JWR2k.mjs} +1 -1
- package/dist/{BaseRealtimeProvider-C2mRn1V_.cjs → BaseRealtimeProvider-ZfFzijdr.cjs} +1 -1
- package/dist/{GeminiLiveProvider-Cc1dkxW6.d.cts → GeminiLiveProvider-C_LQCTX5.d.mts} +1 -1
- package/dist/{GeminiLiveProvider-TvY_cQQZ.d.mts → GeminiLiveProvider-DM-GwD0w.d.cts} +1 -1
- package/dist/{InMemorySessionStore-B5_rq61L.d.cts → InMemorySessionStore-BlsyPnIp.d.cts} +12 -6
- package/dist/{InMemorySessionStore-B5_rq61L.d.mts → InMemorySessionStore-BlsyPnIp.d.mts} +12 -6
- package/dist/{OpenAICompatibleProvider-D-2OOVBU.mjs → OpenAICompatibleProvider-Dqje-boq.mjs} +3 -3
- package/dist/{OpenAICompatibleProvider-Mp0Mefbh.cjs → OpenAICompatibleProvider-j2Nza2q2.cjs} +8 -2
- package/dist/{events-BylBSBW-.mjs → events-D-pvCH_F.mjs} +2 -1
- package/dist/{events-BxDTIKKq.cjs → events-DWX9535M.cjs} +2 -1
- package/dist/gemini.cjs +1 -1
- package/dist/gemini.d.cts +2 -2
- package/dist/gemini.d.mts +2 -2
- package/dist/gemini.mjs +1 -1
- package/dist/gpt-live.cjs +954 -0
- package/dist/gpt-live.d.cts +261 -0
- package/dist/gpt-live.d.mts +261 -0
- package/dist/gpt-live.mjs +936 -0
- package/dist/index.cjs +194 -25
- package/dist/index.d.cts +32 -3
- package/dist/index.d.mts +32 -3
- package/dist/index.mjs +194 -25
- package/dist/openai.cjs +2 -2
- package/dist/openai.d.cts +1 -1
- package/dist/openai.d.mts +1 -1
- package/dist/openai.mjs +2 -2
- package/dist/{rest-BYqiVOhe.mjs → rest-D-DWWVIj.mjs} +1 -1
- package/dist/{rest-BvUKut_k.cjs → rest-OR4VhEif.cjs} +1 -1
- package/dist/{session-config-CbifLlkV.mjs → session-config-C31WZESg.mjs} +1 -1
- package/dist/{session-config-CqJm2Kxz.cjs → session-config-CYjxBqpF.cjs} +1 -1
- package/dist/store.d.cts +2 -2
- package/dist/store.d.mts +2 -2
- package/dist/testing.cjs +376 -0
- package/dist/testing.d.cts +119 -2
- package/dist/testing.d.mts +119 -2
- package/dist/testing.mjs +374 -2
- package/dist/twilio.cjs +1 -1
- package/dist/twilio.mjs +1 -1
- package/dist/xai.cjs +3 -3
- package/dist/xai.d.cts +1 -1
- package/dist/xai.d.mts +1 -1
- package/dist/xai.mjs +3 -3
- package/package.json +14 -2
|
@@ -0,0 +1,954 @@
|
|
|
1
|
+
Object.defineProperty(exports, Symbol.toStringTag, { value: "Module" });
|
|
2
|
+
const require_rolldown_runtime = require("./rolldown-runtime-VH7oDXx4.cjs");
|
|
3
|
+
const require_BaseRealtimeProvider = require("./BaseRealtimeProvider-ZfFzijdr.cjs");
|
|
4
|
+
const require_mulaw = require("./mulaw-DLUObjdP.cjs");
|
|
5
|
+
const require_env = require("./env-DSnGaERV.cjs");
|
|
6
|
+
const require_OpenAICompatibleProvider = require("./OpenAICompatibleProvider-j2Nza2q2.cjs");
|
|
7
|
+
let ws = require("ws");
|
|
8
|
+
ws = require_rolldown_runtime.__toESM(ws, 1);
|
|
9
|
+
//#region src/providers/gpt-live/session-config.ts
|
|
10
|
+
/**
|
|
11
|
+
* `session.start` construction for GPT-Live (`/v1/live/sessions`).
|
|
12
|
+
*
|
|
13
|
+
* The session object is a STRICT schema: an unknown field rejects the whole
|
|
14
|
+
* start with `unknown_parameter` (field-verified Sept 2026), so every key
|
|
15
|
+
* here is a documented one and `providerOptions` / `extraSessionOptions`
|
|
16
|
+
* are deep-merged last as a deliberate escape hatch. Wire facts baked in:
|
|
17
|
+
* - one audio format for both directions, `{ type: 'audio/pcmu', rate: 8000 }`
|
|
18
|
+
* (Twilio's wire format — no transcoding)
|
|
19
|
+
* - instructions, voice, format and history are immutable after start
|
|
20
|
+
* - tools do not live on the voice model: they are `delegation.responses.tools`
|
|
21
|
+
* for the backend Responses model that reasons and calls them
|
|
22
|
+
* - history is `input`: ≤ 128 text messages / ≤ 8,192 rendered tokens
|
|
23
|
+
*/
|
|
24
|
+
const GPT_LIVE_DEFAULT_BACKEND_MODEL = "gpt-5.6-terra";
|
|
25
|
+
const GPT_LIVE_AUDIO_FORMAT = {
|
|
26
|
+
type: "audio/pcmu",
|
|
27
|
+
rate: 8e3
|
|
28
|
+
};
|
|
29
|
+
const HISTORY_MAX_MESSAGES = 128;
|
|
30
|
+
/** ≈ 8,192 tokens at a conservative 2.5 chars/token (Hebrew and other dense scripts). */
|
|
31
|
+
const HISTORY_MAX_CHARS = 2e4;
|
|
32
|
+
function toBackendTool(tool) {
|
|
33
|
+
return {
|
|
34
|
+
type: "function",
|
|
35
|
+
name: tool.name,
|
|
36
|
+
description: tool.description ?? "",
|
|
37
|
+
parameters: tool.parameters
|
|
38
|
+
};
|
|
39
|
+
}
|
|
40
|
+
function buildDelegation(tools, options = {}) {
|
|
41
|
+
const responses = {
|
|
42
|
+
model: options.model ?? "gpt-5.6-terra",
|
|
43
|
+
tools: [...(tools ?? []).map(toBackendTool), ...options.extraTools ?? []],
|
|
44
|
+
tool_choice: options.toolChoice ?? "auto"
|
|
45
|
+
};
|
|
46
|
+
if (options.instructions !== void 0) responses.instructions = options.instructions;
|
|
47
|
+
if (options.parallelToolCalls !== void 0) responses.parallel_tool_calls = options.parallelToolCalls;
|
|
48
|
+
if (options.maxOutputTokens !== void 0) responses.max_output_tokens = options.maxOutputTokens;
|
|
49
|
+
if (options.reasoning !== void 0) responses.reasoning = options.reasoning;
|
|
50
|
+
if (options.serviceTier !== void 0) responses.service_tier = options.serviceTier;
|
|
51
|
+
if (options.text !== void 0) responses.text = options.text;
|
|
52
|
+
return {
|
|
53
|
+
type: "responses",
|
|
54
|
+
responses
|
|
55
|
+
};
|
|
56
|
+
}
|
|
57
|
+
/**
|
|
58
|
+
* History → `input` items, newest-first trimming to the API bounds. A leading
|
|
59
|
+
* `developer` entry (the engine's continuation note) is pinned so trimming
|
|
60
|
+
* never drops the context that explains the rest.
|
|
61
|
+
*/
|
|
62
|
+
function buildHistoryItems(history, limits = {}) {
|
|
63
|
+
if (!history || history.length === 0) return {
|
|
64
|
+
items: [],
|
|
65
|
+
dropped: 0
|
|
66
|
+
};
|
|
67
|
+
const maxMessages = limits.maxMessages ?? HISTORY_MAX_MESSAGES;
|
|
68
|
+
const maxChars = limits.maxChars ?? HISTORY_MAX_CHARS;
|
|
69
|
+
const pinned = history[0].role === "developer" ? [history[0]] : [];
|
|
70
|
+
const rest = history.slice(pinned.length);
|
|
71
|
+
let chars = pinned.reduce((n, e) => n + e.text.length, 0);
|
|
72
|
+
const kept = [];
|
|
73
|
+
for (let i = rest.length - 1; i >= 0; i--) {
|
|
74
|
+
const entry = rest[i];
|
|
75
|
+
if (kept.length + pinned.length >= maxMessages) break;
|
|
76
|
+
if (chars + entry.text.length > maxChars) break;
|
|
77
|
+
chars += entry.text.length;
|
|
78
|
+
kept.unshift(entry);
|
|
79
|
+
}
|
|
80
|
+
return {
|
|
81
|
+
items: [...pinned, ...kept].map((entry) => ({
|
|
82
|
+
type: "message",
|
|
83
|
+
role: entry.role,
|
|
84
|
+
content: [{
|
|
85
|
+
type: entry.role === "assistant" ? "output_text" : "input_text",
|
|
86
|
+
text: entry.text
|
|
87
|
+
}]
|
|
88
|
+
})),
|
|
89
|
+
dropped: rest.length - kept.length
|
|
90
|
+
};
|
|
91
|
+
}
|
|
92
|
+
function buildSessionStart(init, config, eventId, model) {
|
|
93
|
+
const { items, dropped } = buildHistoryItems(init.history, {
|
|
94
|
+
maxMessages: config.historyMaxMessages,
|
|
95
|
+
maxChars: config.historyMaxChars
|
|
96
|
+
});
|
|
97
|
+
const voice = init.voice ?? config.defaultVoice;
|
|
98
|
+
let session = {
|
|
99
|
+
model,
|
|
100
|
+
instructions: init.instructions,
|
|
101
|
+
audio: {
|
|
102
|
+
format: { ...GPT_LIVE_AUDIO_FORMAT },
|
|
103
|
+
...voice ? { output: { voice } } : {}
|
|
104
|
+
},
|
|
105
|
+
delegation: buildDelegation(init.tools, config.delegation),
|
|
106
|
+
...items.length ? { input: items } : {},
|
|
107
|
+
...config.store ? { store: true } : {}
|
|
108
|
+
};
|
|
109
|
+
session = require_BaseRealtimeProvider.deepMerge(session, init.providerOptions);
|
|
110
|
+
session = require_BaseRealtimeProvider.deepMerge(session, config.extraSessionOptions);
|
|
111
|
+
return {
|
|
112
|
+
frame: {
|
|
113
|
+
type: "session.start",
|
|
114
|
+
event_id: eventId,
|
|
115
|
+
session
|
|
116
|
+
},
|
|
117
|
+
droppedHistory: dropped
|
|
118
|
+
};
|
|
119
|
+
}
|
|
120
|
+
//#endregion
|
|
121
|
+
//#region src/providers/gpt-live/speech-gate.ts
|
|
122
|
+
/**
|
|
123
|
+
* Speech gate for full-duplex output streams.
|
|
124
|
+
*
|
|
125
|
+
* GPT-Live streams assistant audio continuously and in real time — silence
|
|
126
|
+
* included (≈100 ms μ-law deltas, digital 0xFF while the model is quiet) —
|
|
127
|
+
* and the wire has no response boundaries: no response.created, no
|
|
128
|
+
* output_audio.done. The engine's playback tracking, tool timing and hangup
|
|
129
|
+
* watchdog all key off `responseStarted` / `responseDone`, so this gate
|
|
130
|
+
* synthesizes utterances from the audio itself: one opens on the first loud
|
|
131
|
+
* 20 ms sub-frame and closes after `quietMs` of quiet AUDIO. Quiet is
|
|
132
|
+
* measured on the stream's own clock (bytes received), never wall-clock, so
|
|
133
|
+
* a stalled socket cannot fake a finished utterance and tests can script
|
|
134
|
+
* silence instantly. Field numbers (Sept 2026, 8 kHz μ-law): quiet frames
|
|
135
|
+
* p99 ≈ −50 dBFS, speech p05 ≈ −44 dBFS; LiveKit and Pipecat both close at
|
|
136
|
+
* 0.8 s.
|
|
137
|
+
*/
|
|
138
|
+
const DEFAULT_GATE_THRESHOLD_RMS = .004;
|
|
139
|
+
const DEFAULT_GATE_QUIET_MS = 800;
|
|
140
|
+
const SUBFRAME_BYTES = 160;
|
|
141
|
+
const BYTES_PER_MS = 8;
|
|
142
|
+
var SpeechGate = class {
|
|
143
|
+
threshold;
|
|
144
|
+
quietLimitMs;
|
|
145
|
+
open = false;
|
|
146
|
+
quietMs = 0;
|
|
147
|
+
utteranceMs = 0;
|
|
148
|
+
constructor(options = {}) {
|
|
149
|
+
this.threshold = options.thresholdRms ?? .004;
|
|
150
|
+
this.quietLimitMs = options.quietMs ?? 800;
|
|
151
|
+
}
|
|
152
|
+
get isOpen() {
|
|
153
|
+
return this.open;
|
|
154
|
+
}
|
|
155
|
+
/** Feed one output delta; returns the transitions it caused, in order. */
|
|
156
|
+
feed(bytes) {
|
|
157
|
+
const events = [];
|
|
158
|
+
for (let offset = 0; offset < bytes.length; offset += SUBFRAME_BYTES) {
|
|
159
|
+
const frame = bytes.subarray(offset, Math.min(offset + SUBFRAME_BYTES, bytes.length));
|
|
160
|
+
const frameMs = frame.length / BYTES_PER_MS;
|
|
161
|
+
if (isSpeech(frame, this.threshold)) {
|
|
162
|
+
if (!this.open) {
|
|
163
|
+
this.open = true;
|
|
164
|
+
this.utteranceMs = 0;
|
|
165
|
+
events.push({ type: "open" });
|
|
166
|
+
}
|
|
167
|
+
this.quietMs = 0;
|
|
168
|
+
this.utteranceMs += frameMs;
|
|
169
|
+
} else if (this.open) {
|
|
170
|
+
this.quietMs += frameMs;
|
|
171
|
+
this.utteranceMs += frameMs;
|
|
172
|
+
if (this.quietMs >= this.quietLimitMs) {
|
|
173
|
+
events.push({
|
|
174
|
+
type: "close",
|
|
175
|
+
utteranceMs: this.utteranceMs - this.quietMs
|
|
176
|
+
});
|
|
177
|
+
this.open = false;
|
|
178
|
+
this.quietMs = 0;
|
|
179
|
+
this.utteranceMs = 0;
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
}
|
|
183
|
+
return events;
|
|
184
|
+
}
|
|
185
|
+
/** Force-close an open utterance (stream stalled, session closing). */
|
|
186
|
+
close() {
|
|
187
|
+
if (!this.open) return null;
|
|
188
|
+
const event = {
|
|
189
|
+
type: "close",
|
|
190
|
+
utteranceMs: this.utteranceMs - this.quietMs
|
|
191
|
+
};
|
|
192
|
+
this.open = false;
|
|
193
|
+
this.quietMs = 0;
|
|
194
|
+
this.utteranceMs = 0;
|
|
195
|
+
return event;
|
|
196
|
+
}
|
|
197
|
+
};
|
|
198
|
+
/** μ-law digital zero is 0xFF (some encoders emit 0x7F for negative zero). */
|
|
199
|
+
function isDigitalSilence(bytes) {
|
|
200
|
+
for (let i = 0; i < bytes.length; i++) {
|
|
201
|
+
const b = bytes[i];
|
|
202
|
+
if (b !== 255 && b !== 127) return false;
|
|
203
|
+
}
|
|
204
|
+
return true;
|
|
205
|
+
}
|
|
206
|
+
function isSpeech(frame, threshold) {
|
|
207
|
+
if (frame.length === 0 || isDigitalSilence(frame)) return false;
|
|
208
|
+
const pcm = require_mulaw.mulawToPcm16(frame);
|
|
209
|
+
let acc = 0;
|
|
210
|
+
for (let i = 0; i < pcm.length; i++) {
|
|
211
|
+
const v = pcm[i] / 32768;
|
|
212
|
+
acc += v * v;
|
|
213
|
+
}
|
|
214
|
+
return Math.sqrt(acc / pcm.length) >= threshold;
|
|
215
|
+
}
|
|
216
|
+
//#endregion
|
|
217
|
+
//#region src/providers/gpt-live/transcript-grouper.ts
|
|
218
|
+
var TranscriptGrouper = class {
|
|
219
|
+
options;
|
|
220
|
+
current = null;
|
|
221
|
+
idleTimer = null;
|
|
222
|
+
constructor(options) {
|
|
223
|
+
this.options = options;
|
|
224
|
+
}
|
|
225
|
+
push(delta, startMs, endMs) {
|
|
226
|
+
if (this.current && startMs - this.current.endMs > this.options.gapMs) this.flush();
|
|
227
|
+
if (!this.current) this.current = {
|
|
228
|
+
text: "",
|
|
229
|
+
startMs,
|
|
230
|
+
endMs,
|
|
231
|
+
tag: this.options.tagFor?.()
|
|
232
|
+
};
|
|
233
|
+
this.current.text += delta;
|
|
234
|
+
this.current.endMs = Math.max(this.current.endMs, endMs);
|
|
235
|
+
this.armIdle();
|
|
236
|
+
}
|
|
237
|
+
/** Close the open turn now (session ending, utterance boundary reached). */
|
|
238
|
+
flush() {
|
|
239
|
+
this.clearIdle();
|
|
240
|
+
const turn = this.current;
|
|
241
|
+
this.current = null;
|
|
242
|
+
if (!turn) return;
|
|
243
|
+
const text = turn.text.trim();
|
|
244
|
+
if (text.length === 0) return;
|
|
245
|
+
this.options.onTurn({
|
|
246
|
+
...turn,
|
|
247
|
+
text
|
|
248
|
+
});
|
|
249
|
+
}
|
|
250
|
+
dispose() {
|
|
251
|
+
this.clearIdle();
|
|
252
|
+
this.current = null;
|
|
253
|
+
}
|
|
254
|
+
armIdle() {
|
|
255
|
+
this.clearIdle();
|
|
256
|
+
this.idleTimer = setTimeout(() => {
|
|
257
|
+
this.idleTimer = null;
|
|
258
|
+
this.flush();
|
|
259
|
+
}, this.options.idleMs);
|
|
260
|
+
this.idleTimer.unref?.();
|
|
261
|
+
}
|
|
262
|
+
clearIdle() {
|
|
263
|
+
if (this.idleTimer) {
|
|
264
|
+
clearTimeout(this.idleTimer);
|
|
265
|
+
this.idleTimer = null;
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
};
|
|
269
|
+
//#endregion
|
|
270
|
+
//#region src/providers/gpt-live/GptLiveProvider.ts
|
|
271
|
+
/**
|
|
272
|
+
* GPT-Live provider (`/v1/live/sessions`, raw WebSocket) — OpenAI's
|
|
273
|
+
* full-duplex speech-to-speech model. A different API from Realtime, not a
|
|
274
|
+
* new Realtime model: its own handshake (`session.start` → `session.started`),
|
|
275
|
+
* its own events, and a different division of labour that this provider
|
|
276
|
+
* translates into the engine's normalized contract:
|
|
277
|
+
*
|
|
278
|
+
* - **Turn-taking is the model's.** It listens while it speaks and stops on
|
|
279
|
+
* its own when the caller talks. There is no VAD config, no cancel, no
|
|
280
|
+
* truncate, no speech_started/stopped; `capabilities.turnTaking: 'model'`
|
|
281
|
+
* tells the engine to keep its hands off (see parity tests).
|
|
282
|
+
* - **Output is a continuous real-time stream, silence included** (100 ms
|
|
283
|
+
* μ-law deltas, digital 0xFF when quiet). Response boundaries are
|
|
284
|
+
* synthesized by a speech gate on the audio itself (`speech-gate.ts`);
|
|
285
|
+
* silence between utterances is not forwarded.
|
|
286
|
+
* - **The backend does the thinking.** Tools are declared on a Responses
|
|
287
|
+
* model (`delegation.responses`), calls arrive wrapped in `response.event`
|
|
288
|
+
* envelopes, results go back as `response.item.create` + one
|
|
289
|
+
* `response.create`. The voice keeps talking meanwhile
|
|
290
|
+
* (`capabilities.decoupledBackend`).
|
|
291
|
+
* - **Text reaches the model only through appends** (≤ 500 tokens each):
|
|
292
|
+
* instructions (policy), thinking (quiet context), commentary (say this).
|
|
293
|
+
* `createResponse({ instructions })` maps to commentary — field-tested as
|
|
294
|
+
* the only append that reliably produces speech on demand (Sept 2026).
|
|
295
|
+
* - **Instructions, voice and history are immutable after start** — handoffs
|
|
296
|
+
* reconnect and seed the attributed transcript via `session.input`
|
|
297
|
+
* (`capabilities.startupHistory`).
|
|
298
|
+
* - **Billing is per second of session**, reported as cumulative
|
|
299
|
+
* `session.usage.updated` ticks; `close()` sends `session.close` and waits
|
|
300
|
+
* for `session.closed` so the final usage is confirmed.
|
|
301
|
+
*/
|
|
302
|
+
const GPT_LIVE_DEFAULT_BASE_URL = "wss://api.openai.com/v1/live/sessions";
|
|
303
|
+
const NON_RETRIABLE_CLOSE_CODES = /* @__PURE__ */ new Set([
|
|
304
|
+
1002,
|
|
305
|
+
1003,
|
|
306
|
+
1007,
|
|
307
|
+
1008
|
|
308
|
+
]);
|
|
309
|
+
/** Appends are capped at 500 tokens; split conservatively (dense scripts run ~2.5 chars/token). */
|
|
310
|
+
const APPEND_MAX_CHARS = 1200;
|
|
311
|
+
const TRANSCRIPT_IDLE_EXTRA_MS = 300;
|
|
312
|
+
const GATE_STALL_EXTRA_MS = 500;
|
|
313
|
+
const ACK_TYPES = {
|
|
314
|
+
"session.instructions.append": "session.instructions.appended",
|
|
315
|
+
"session.thinking.append": "session.thinking.appended",
|
|
316
|
+
"session.commentary.append": "session.commentary.appended",
|
|
317
|
+
"session.update": "session.updated",
|
|
318
|
+
"session.input_audio.mute": "session.input_audio.muted",
|
|
319
|
+
"session.input_audio.unmute": "session.input_audio.unmuted"
|
|
320
|
+
};
|
|
321
|
+
let eventSeq = 0;
|
|
322
|
+
const nextEventId = (prefix) => `${prefix}_${++eventSeq}`;
|
|
323
|
+
var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimeProvider {
|
|
324
|
+
name;
|
|
325
|
+
capabilities = {
|
|
326
|
+
truncate: false,
|
|
327
|
+
sessionUpdate: false,
|
|
328
|
+
voiceChangeMidSession: false,
|
|
329
|
+
transcodeRequired: false,
|
|
330
|
+
resumption: false,
|
|
331
|
+
agentTranscriptDeltas: true,
|
|
332
|
+
vadInterruptControl: false,
|
|
333
|
+
turnTaking: "model",
|
|
334
|
+
startupHistory: true,
|
|
335
|
+
decoupledBackend: true
|
|
336
|
+
};
|
|
337
|
+
config;
|
|
338
|
+
logger;
|
|
339
|
+
ws = null;
|
|
340
|
+
ready = false;
|
|
341
|
+
intentionalClose = false;
|
|
342
|
+
sessionInit = null;
|
|
343
|
+
/** Server-assigned session id (sideband attach, forking, recording download). */
|
|
344
|
+
sessionId = null;
|
|
345
|
+
/** Unix seconds at which the server expires the session (120 min at launch). */
|
|
346
|
+
expiresAt = null;
|
|
347
|
+
gate;
|
|
348
|
+
gateStallTimer = null;
|
|
349
|
+
utteranceCounter = 0;
|
|
350
|
+
currentUtteranceId = null;
|
|
351
|
+
lastUtteranceId = null;
|
|
352
|
+
inputTranscript;
|
|
353
|
+
outputTranscript;
|
|
354
|
+
acks = /* @__PURE__ */ new Map();
|
|
355
|
+
seenCalls = /* @__PURE__ */ new Set();
|
|
356
|
+
sessionClosed = null;
|
|
357
|
+
closeListeners = [];
|
|
358
|
+
constructor(config, logger = require_BaseRealtimeProvider.noopLogger) {
|
|
359
|
+
super();
|
|
360
|
+
this.config = config;
|
|
361
|
+
this.logger = logger;
|
|
362
|
+
this.name = config.providerName ?? "gpt-live";
|
|
363
|
+
this.gate = new SpeechGate(config.speechGate);
|
|
364
|
+
const gapMs = config.transcriptGapMs ?? 800;
|
|
365
|
+
this.inputTranscript = new TranscriptGrouper({
|
|
366
|
+
gapMs,
|
|
367
|
+
idleMs: gapMs + TRANSCRIPT_IDLE_EXTRA_MS,
|
|
368
|
+
onTurn: (turn) => this.emit("userTranscript", { text: turn.text })
|
|
369
|
+
});
|
|
370
|
+
this.outputTranscript = new TranscriptGrouper({
|
|
371
|
+
gapMs,
|
|
372
|
+
idleMs: gapMs + TRANSCRIPT_IDLE_EXTRA_MS,
|
|
373
|
+
tagFor: () => this.currentUtteranceId ?? this.lastUtteranceId ?? void 0,
|
|
374
|
+
onTurn: (turn) => this.emit("agentTranscript", {
|
|
375
|
+
responseId: turn.tag ?? "",
|
|
376
|
+
text: turn.text
|
|
377
|
+
})
|
|
378
|
+
});
|
|
379
|
+
}
|
|
380
|
+
get isConnected() {
|
|
381
|
+
return this.ready && this.ws?.readyState === ws.default.OPEN;
|
|
382
|
+
}
|
|
383
|
+
async connect(init) {
|
|
384
|
+
if (this.ws) await this.close();
|
|
385
|
+
this.sessionInit = init;
|
|
386
|
+
this.intentionalClose = false;
|
|
387
|
+
this.ready = false;
|
|
388
|
+
this.sessionClosed = null;
|
|
389
|
+
this.sessionId = null;
|
|
390
|
+
this.expiresAt = null;
|
|
391
|
+
this.resetUtteranceState();
|
|
392
|
+
if (init.vad !== void 0 || init.transcription !== void 0 || init.temperature !== void 0) this.logger.debug("gpt-live: vad/transcription/temperature are model-owned and ignored");
|
|
393
|
+
const startId = nextEventId("start");
|
|
394
|
+
const { frame, droppedHistory } = buildSessionStart(init, this.sessionConfig(), startId, this.config.model);
|
|
395
|
+
if (droppedHistory > 0) this.logger.warn("gpt-live: startup history trimmed to the API bounds", { dropped: droppedHistory });
|
|
396
|
+
const ws$1 = new ws.default(this.config.baseUrl ?? "wss://api.openai.com/v1/live/sessions", {
|
|
397
|
+
headers: {
|
|
398
|
+
Authorization: `Bearer ${this.config.apiKey}`,
|
|
399
|
+
...this.config.headers
|
|
400
|
+
},
|
|
401
|
+
perMessageDeflate: false
|
|
402
|
+
});
|
|
403
|
+
this.ws = ws$1;
|
|
404
|
+
await new Promise((resolve, reject) => {
|
|
405
|
+
const timeoutMs = this.config.connectTimeoutMs ?? 15e3;
|
|
406
|
+
const timer = setTimeout(() => fail(/* @__PURE__ */ new Error(`${this.name} session not started within ${timeoutMs}ms`)), timeoutMs);
|
|
407
|
+
timer.unref?.();
|
|
408
|
+
let settled = false;
|
|
409
|
+
const fail = (error) => {
|
|
410
|
+
if (settled) return;
|
|
411
|
+
settled = true;
|
|
412
|
+
clearTimeout(timer);
|
|
413
|
+
try {
|
|
414
|
+
ws$1.close();
|
|
415
|
+
} catch {}
|
|
416
|
+
reject(error);
|
|
417
|
+
};
|
|
418
|
+
ws$1.on("unexpected-response", (_request, response) => {
|
|
419
|
+
let body = "";
|
|
420
|
+
response.setEncoding("utf8");
|
|
421
|
+
response.on("data", (chunk) => {
|
|
422
|
+
if (body.length < 512) body += chunk;
|
|
423
|
+
});
|
|
424
|
+
response.on("end", () => {
|
|
425
|
+
const detail = body.trim().slice(0, 500);
|
|
426
|
+
this.logger.error("provider rejected the WebSocket upgrade", {
|
|
427
|
+
status: response.statusCode,
|
|
428
|
+
body: detail
|
|
429
|
+
});
|
|
430
|
+
fail(/* @__PURE__ */ new Error(`${this.name} rejected the WebSocket upgrade: HTTP ${response.statusCode}${detail ? ` — ${detail}` : ""}`));
|
|
431
|
+
});
|
|
432
|
+
});
|
|
433
|
+
ws$1.on("open", () => {
|
|
434
|
+
this.emit("open");
|
|
435
|
+
this.send(frame);
|
|
436
|
+
});
|
|
437
|
+
ws$1.on("message", (raw) => {
|
|
438
|
+
const event = this.parseEvent(raw);
|
|
439
|
+
if (!event) return;
|
|
440
|
+
if (!settled) {
|
|
441
|
+
if (event.type === "session.started") {
|
|
442
|
+
settled = true;
|
|
443
|
+
clearTimeout(timer);
|
|
444
|
+
this.sessionId = event.session?.id ?? null;
|
|
445
|
+
this.expiresAt = typeof event.session?.expires_at === "number" ? event.session.expires_at : null;
|
|
446
|
+
this.ready = true;
|
|
447
|
+
this.logger.info("gpt-live session started", {
|
|
448
|
+
sessionId: this.sessionId,
|
|
449
|
+
expiresAt: this.expiresAt ? (/* @__PURE__ */ new Date(this.expiresAt * 1e3)).toISOString() : void 0
|
|
450
|
+
});
|
|
451
|
+
resolve();
|
|
452
|
+
return;
|
|
453
|
+
}
|
|
454
|
+
if (event.type === "error") {
|
|
455
|
+
fail(/* @__PURE__ */ new Error(`${this.name} session.start rejected: ${JSON.stringify(event.error ?? event)}`));
|
|
456
|
+
return;
|
|
457
|
+
}
|
|
458
|
+
return;
|
|
459
|
+
}
|
|
460
|
+
this.handleEvent(event);
|
|
461
|
+
});
|
|
462
|
+
ws$1.on("error", (error) => {
|
|
463
|
+
const err = error instanceof Error ? error : new Error(String(error));
|
|
464
|
+
if (!settled) fail(err);
|
|
465
|
+
else this.emit("error", err);
|
|
466
|
+
});
|
|
467
|
+
ws$1.on("close", (code, reasonBuf) => {
|
|
468
|
+
const reason = reasonBuf?.toString();
|
|
469
|
+
this.ready = false;
|
|
470
|
+
this.failPendingAcks();
|
|
471
|
+
this.finishUtteranceOnClose();
|
|
472
|
+
const listeners = this.closeListeners;
|
|
473
|
+
this.closeListeners = [];
|
|
474
|
+
for (const listener of listeners) listener();
|
|
475
|
+
if (!settled) {
|
|
476
|
+
fail(/* @__PURE__ */ new Error(`${this.name} socket closed during setup (${code} ${reason ?? ""})`));
|
|
477
|
+
return;
|
|
478
|
+
}
|
|
479
|
+
this.emit("close", {
|
|
480
|
+
code,
|
|
481
|
+
reason,
|
|
482
|
+
retriable: this.isRetriableClose(code)
|
|
483
|
+
});
|
|
484
|
+
});
|
|
485
|
+
});
|
|
486
|
+
}
|
|
487
|
+
async close() {
|
|
488
|
+
this.intentionalClose = true;
|
|
489
|
+
this.ready = false;
|
|
490
|
+
const ws$2 = this.ws;
|
|
491
|
+
this.ws = null;
|
|
492
|
+
this.inputTranscript.dispose();
|
|
493
|
+
this.outputTranscript.dispose();
|
|
494
|
+
this.clearGateStall();
|
|
495
|
+
if (!ws$2 || ws$2.readyState === ws.default.CLOSED) return;
|
|
496
|
+
await new Promise((resolve) => {
|
|
497
|
+
const timer = setTimeout(() => finish(), this.config.closeTimeoutMs ?? 3e3);
|
|
498
|
+
timer.unref?.();
|
|
499
|
+
let done = false;
|
|
500
|
+
const finish = () => {
|
|
501
|
+
if (done) return;
|
|
502
|
+
done = true;
|
|
503
|
+
clearTimeout(timer);
|
|
504
|
+
try {
|
|
505
|
+
if (ws$2.readyState === ws.default.OPEN || ws$2.readyState === ws.default.CONNECTING) ws$2.close(1e3);
|
|
506
|
+
} catch {}
|
|
507
|
+
setTimeout(() => {
|
|
508
|
+
try {
|
|
509
|
+
ws$2.terminate();
|
|
510
|
+
} catch {}
|
|
511
|
+
}, 1e3).unref?.();
|
|
512
|
+
resolve();
|
|
513
|
+
};
|
|
514
|
+
this.closeListeners.push(finish);
|
|
515
|
+
if (ws$2.readyState === ws.default.OPEN) try {
|
|
516
|
+
ws$2.send(JSON.stringify({
|
|
517
|
+
type: "session.close",
|
|
518
|
+
event_id: nextEventId("close")
|
|
519
|
+
}));
|
|
520
|
+
} catch {
|
|
521
|
+
finish();
|
|
522
|
+
}
|
|
523
|
+
else finish();
|
|
524
|
+
});
|
|
525
|
+
}
|
|
526
|
+
sendAudio(base64Mulaw) {
|
|
527
|
+
if (!this.isConnected) return;
|
|
528
|
+
this.send({
|
|
529
|
+
type: "session.input_audio.append",
|
|
530
|
+
audio: base64Mulaw
|
|
531
|
+
});
|
|
532
|
+
}
|
|
533
|
+
/**
|
|
534
|
+
* Text has three doors into a GPT-Live session, by intent rather than by
|
|
535
|
+
* conversation role: `system` → instructions (behaviour), `user` → thinking
|
|
536
|
+
* (quiet context: keypad entries, deferred results — the model decides how
|
|
537
|
+
* to react), `assistant` → thinking as well, worded as something already
|
|
538
|
+
* said (commentary would make the model say it again).
|
|
539
|
+
*/
|
|
540
|
+
sendText(text, options = {}) {
|
|
541
|
+
const role = options.role ?? "user";
|
|
542
|
+
if (role === "system") this.append("session.instructions.append", text);
|
|
543
|
+
else if (role === "assistant") this.append("session.thinking.append", `You already said this to the caller earlier: "${text}"`);
|
|
544
|
+
else this.append("session.thinking.append", text);
|
|
545
|
+
}
|
|
546
|
+
sendToolResult(callId, output, options = {}) {
|
|
547
|
+
this.send({
|
|
548
|
+
type: "response.item.create",
|
|
549
|
+
event_id: nextEventId("tool"),
|
|
550
|
+
item: {
|
|
551
|
+
type: "function_call_output",
|
|
552
|
+
call_id: callId,
|
|
553
|
+
output: typeof output === "string" ? output : JSON.stringify(output ?? null)
|
|
554
|
+
}
|
|
555
|
+
});
|
|
556
|
+
if (options.triggerResponse !== false) this.send({
|
|
557
|
+
type: "response.create",
|
|
558
|
+
event_id: nextEventId("continue")
|
|
559
|
+
});
|
|
560
|
+
}
|
|
561
|
+
/**
|
|
562
|
+
* "Speak now" has no direct verb on this API. Commentary — information the
|
|
563
|
+
* model should say aloud, paraphrasing allowed — is the append that
|
|
564
|
+
* produced speech on demand in every field test, greeting and goodbye
|
|
565
|
+
* alike; an instruction-shaped text is followed rather than read out.
|
|
566
|
+
*/
|
|
567
|
+
createResponse(options = {}) {
|
|
568
|
+
if (options.instructions) this.append("session.commentary.append", options.instructions);
|
|
569
|
+
else this.append("session.instructions.append", "Respond to the caller now, without waiting for them to speak.");
|
|
570
|
+
}
|
|
571
|
+
/**
|
|
572
|
+
* Only the backend delegation can change mid-session (`session.update` is
|
|
573
|
+
* sparse and limited to `delegation.responses`): tool lists land there;
|
|
574
|
+
* instructions / voice / vad are immutable and are reported, not applied.
|
|
575
|
+
*/
|
|
576
|
+
async updateSession(patch, options = {}) {
|
|
577
|
+
if (!this.sessionInit) throw new Error("updateSession before connect");
|
|
578
|
+
const ignored = Object.keys(patch).filter((key) => key !== "tools" && key !== "providerOptions");
|
|
579
|
+
if (ignored.length > 0) this.logger.warn("gpt-live: session fields are immutable after start — ignored", { fields: ignored });
|
|
580
|
+
this.sessionInit = {
|
|
581
|
+
...this.sessionInit,
|
|
582
|
+
...patch
|
|
583
|
+
};
|
|
584
|
+
let responses = {};
|
|
585
|
+
if (patch.tools) responses.tools = [...patch.tools.map(toBackendTool), ...this.config.delegation?.extraTools ?? []];
|
|
586
|
+
const nativeResponses = patch.providerOptions?.delegation?.responses;
|
|
587
|
+
if (nativeResponses && typeof nativeResponses === "object") responses = require_BaseRealtimeProvider.deepMerge(responses, nativeResponses);
|
|
588
|
+
if (Object.keys(responses).length === 0) return false;
|
|
589
|
+
if (!this.isConnected) return false;
|
|
590
|
+
const eventId = nextEventId("update");
|
|
591
|
+
const acked = this.awaitAck(eventId, ACK_TYPES["session.update"]);
|
|
592
|
+
this.send({
|
|
593
|
+
type: "session.update",
|
|
594
|
+
event_id: eventId,
|
|
595
|
+
session: { delegation: {
|
|
596
|
+
type: "responses",
|
|
597
|
+
responses
|
|
598
|
+
} }
|
|
599
|
+
});
|
|
600
|
+
return options.awaitAck ? acked : true;
|
|
601
|
+
}
|
|
602
|
+
handleEvent(event) {
|
|
603
|
+
switch (event.type) {
|
|
604
|
+
case "session.output_audio.delta":
|
|
605
|
+
if (typeof event.delta !== "string" || event.delta.length === 0) break;
|
|
606
|
+
this.handleOutputAudio(event.delta);
|
|
607
|
+
break;
|
|
608
|
+
case "session.input_transcript.delta":
|
|
609
|
+
if (typeof event.delta === "string") this.inputTranscript.push(event.delta, numberOr(event.start_ms, 0), numberOr(event.end_ms, 0));
|
|
610
|
+
break;
|
|
611
|
+
case "session.output_transcript.delta":
|
|
612
|
+
if (typeof event.delta === "string") {
|
|
613
|
+
this.emit("agentTranscriptDelta", {
|
|
614
|
+
responseId: this.currentUtteranceId ?? this.lastUtteranceId ?? "",
|
|
615
|
+
delta: event.delta
|
|
616
|
+
});
|
|
617
|
+
this.outputTranscript.push(event.delta, numberOr(event.start_ms, 0), numberOr(event.end_ms, 0));
|
|
618
|
+
}
|
|
619
|
+
break;
|
|
620
|
+
case "session.delegation.created":
|
|
621
|
+
this.logger.debug("gpt-live delegation created", {
|
|
622
|
+
id: event.delegation?.id,
|
|
623
|
+
target: event.delegation?.target,
|
|
624
|
+
responseId: event.delegation?.response_id
|
|
625
|
+
});
|
|
626
|
+
break;
|
|
627
|
+
case "response.event":
|
|
628
|
+
this.handleBackendEvent(event.event ?? {}, event.delegation_id);
|
|
629
|
+
break;
|
|
630
|
+
case "session.usage.updated": {
|
|
631
|
+
const seconds = event.usage?.seconds;
|
|
632
|
+
if (typeof seconds === "number") this.emit("usage", {
|
|
633
|
+
inputTokens: 0,
|
|
634
|
+
outputTokens: 0,
|
|
635
|
+
totalTokens: 0,
|
|
636
|
+
audioSeconds: seconds,
|
|
637
|
+
raw: event
|
|
638
|
+
});
|
|
639
|
+
break;
|
|
640
|
+
}
|
|
641
|
+
case "session.instructions.appended":
|
|
642
|
+
case "session.thinking.appended":
|
|
643
|
+
case "session.commentary.appended":
|
|
644
|
+
case "session.updated":
|
|
645
|
+
case "session.input_audio.muted":
|
|
646
|
+
case "session.input_audio.unmuted":
|
|
647
|
+
this.resolveAck(event.client_event_id, true);
|
|
648
|
+
break;
|
|
649
|
+
case "session.closed": {
|
|
650
|
+
const seconds = event.usage?.seconds;
|
|
651
|
+
this.sessionClosed = {
|
|
652
|
+
reason: event.reason,
|
|
653
|
+
usageSeconds: typeof seconds === "number" ? seconds : void 0
|
|
654
|
+
};
|
|
655
|
+
this.logger.info("gpt-live session closed", {
|
|
656
|
+
reason: event.reason,
|
|
657
|
+
usageSeconds: seconds
|
|
658
|
+
});
|
|
659
|
+
if (typeof seconds === "number") this.emit("usage", {
|
|
660
|
+
inputTokens: 0,
|
|
661
|
+
outputTokens: 0,
|
|
662
|
+
totalTokens: 0,
|
|
663
|
+
audioSeconds: seconds,
|
|
664
|
+
raw: event
|
|
665
|
+
});
|
|
666
|
+
break;
|
|
667
|
+
}
|
|
668
|
+
case "error":
|
|
669
|
+
if (event.error?.client_event_id) this.resolveAck(event.error.client_event_id, false);
|
|
670
|
+
this.logger.warn("provider error event", { error: event.error });
|
|
671
|
+
this.emit("error", /* @__PURE__ */ new Error(`${this.name} error: ${JSON.stringify(event.error ?? event)}`));
|
|
672
|
+
break;
|
|
673
|
+
case "info": this.logger.debug("gpt-live info", {
|
|
674
|
+
code: event.code,
|
|
675
|
+
message: event.message
|
|
676
|
+
});
|
|
677
|
+
}
|
|
678
|
+
}
|
|
679
|
+
handleOutputAudio(delta) {
|
|
680
|
+
const bytes = Buffer.from(delta, "base64");
|
|
681
|
+
const wasOpen = this.gate.isOpen;
|
|
682
|
+
const events = this.gate.feed(bytes);
|
|
683
|
+
let forwardId = wasOpen ? this.currentUtteranceId : null;
|
|
684
|
+
let close = null;
|
|
685
|
+
for (const gateEvent of events) if (gateEvent.type === "open") {
|
|
686
|
+
this.beginUtterance();
|
|
687
|
+
forwardId = this.currentUtteranceId;
|
|
688
|
+
} else close = gateEvent;
|
|
689
|
+
if (forwardId) this.emit("audio", {
|
|
690
|
+
base64Mulaw: delta,
|
|
691
|
+
responseId: forwardId
|
|
692
|
+
});
|
|
693
|
+
if (close) this.endUtterance();
|
|
694
|
+
else if (this.gate.isOpen) this.armGateStall();
|
|
695
|
+
}
|
|
696
|
+
handleBackendEvent(inner, delegationId) {
|
|
697
|
+
const type = inner.type ?? "";
|
|
698
|
+
if (type === "response.output_item.done" && inner.item?.type === "function_call") {
|
|
699
|
+
const item = inner.item;
|
|
700
|
+
const callId = item.call_id ?? item.id ?? `call_${Date.now()}`;
|
|
701
|
+
if (this.seenCalls.has(callId)) return;
|
|
702
|
+
this.seenCalls.add(callId);
|
|
703
|
+
if (this.seenCalls.size > 1e3) this.seenCalls.delete(this.seenCalls.values().next().value);
|
|
704
|
+
this.emit("toolCall", {
|
|
705
|
+
id: callId,
|
|
706
|
+
name: item.name,
|
|
707
|
+
argumentsJson: typeof item.arguments === "string" ? item.arguments : "{}",
|
|
708
|
+
responseId: delegationId,
|
|
709
|
+
itemId: item.id
|
|
710
|
+
});
|
|
711
|
+
return;
|
|
712
|
+
}
|
|
713
|
+
if (type === "response.completed" || type === "response.done") {
|
|
714
|
+
const usage = require_OpenAICompatibleProvider.normalizeUsage(inner.response?.usage);
|
|
715
|
+
if (usage) this.emit("usage", usage);
|
|
716
|
+
return;
|
|
717
|
+
}
|
|
718
|
+
if (type === "response.failed" || type === "response.incomplete") this.logger.warn("gpt-live backend response did not complete", {
|
|
719
|
+
type,
|
|
720
|
+
delegationId,
|
|
721
|
+
error: inner.response?.error
|
|
722
|
+
});
|
|
723
|
+
}
|
|
724
|
+
beginUtterance() {
|
|
725
|
+
this.currentUtteranceId = `live_utt_${++this.utteranceCounter}`;
|
|
726
|
+
this.lastUtteranceId = this.currentUtteranceId;
|
|
727
|
+
this.emit("responseStarted", { responseId: this.currentUtteranceId });
|
|
728
|
+
}
|
|
729
|
+
endUtterance() {
|
|
730
|
+
this.clearGateStall();
|
|
731
|
+
const id = this.currentUtteranceId;
|
|
732
|
+
this.currentUtteranceId = null;
|
|
733
|
+
if (id) this.emit("responseDone", { responseId: id });
|
|
734
|
+
}
|
|
735
|
+
/** The stream is real-time: no delta for a quiet window means it stalled — close the utterance. */
|
|
736
|
+
armGateStall() {
|
|
737
|
+
this.clearGateStall();
|
|
738
|
+
const quietMs = this.config.speechGate?.quietMs ?? 800;
|
|
739
|
+
this.gateStallTimer = setTimeout(() => {
|
|
740
|
+
this.gateStallTimer = null;
|
|
741
|
+
if (this.gate.close()) this.endUtterance();
|
|
742
|
+
}, quietMs + GATE_STALL_EXTRA_MS);
|
|
743
|
+
this.gateStallTimer.unref?.();
|
|
744
|
+
}
|
|
745
|
+
clearGateStall() {
|
|
746
|
+
if (this.gateStallTimer) {
|
|
747
|
+
clearTimeout(this.gateStallTimer);
|
|
748
|
+
this.gateStallTimer = null;
|
|
749
|
+
}
|
|
750
|
+
}
|
|
751
|
+
finishUtteranceOnClose() {
|
|
752
|
+
this.clearGateStall();
|
|
753
|
+
if (this.gate.close()) this.endUtterance();
|
|
754
|
+
this.inputTranscript.flush();
|
|
755
|
+
this.outputTranscript.flush();
|
|
756
|
+
}
|
|
757
|
+
resetUtteranceState() {
|
|
758
|
+
this.clearGateStall();
|
|
759
|
+
this.gate.close();
|
|
760
|
+
this.currentUtteranceId = null;
|
|
761
|
+
this.lastUtteranceId = null;
|
|
762
|
+
this.seenCalls.clear();
|
|
763
|
+
}
|
|
764
|
+
/** Send `content` as one or more appends (500-token cap); resolves with the last chunk's ack. */
|
|
765
|
+
async append(type, content, delegationId = null) {
|
|
766
|
+
if (!this.isConnected) return false;
|
|
767
|
+
const chunks = splitForAppend(content);
|
|
768
|
+
if (chunks.length > 1) this.logger.warn("gpt-live: append split to respect the 500-token cap", {
|
|
769
|
+
type,
|
|
770
|
+
chunks: chunks.length
|
|
771
|
+
});
|
|
772
|
+
let acked = false;
|
|
773
|
+
for (const chunk of chunks) {
|
|
774
|
+
const eventId = nextEventId("append");
|
|
775
|
+
const ack = this.awaitAck(eventId, ACK_TYPES[type]);
|
|
776
|
+
this.send({
|
|
777
|
+
type,
|
|
778
|
+
event_id: eventId,
|
|
779
|
+
delegation_id: delegationId,
|
|
780
|
+
content: chunk
|
|
781
|
+
});
|
|
782
|
+
acked = await ack;
|
|
783
|
+
if (!acked) break;
|
|
784
|
+
}
|
|
785
|
+
return acked;
|
|
786
|
+
}
|
|
787
|
+
awaitAck(eventId, ackType) {
|
|
788
|
+
return new Promise((resolve) => {
|
|
789
|
+
const timer = setTimeout(() => {
|
|
790
|
+
if (this.acks.delete(eventId)) {
|
|
791
|
+
this.logger.warn("gpt-live: no acknowledgement for client event", {
|
|
792
|
+
eventId,
|
|
793
|
+
ackType
|
|
794
|
+
});
|
|
795
|
+
resolve(false);
|
|
796
|
+
}
|
|
797
|
+
}, this.config.ackTimeoutMs ?? 5e3);
|
|
798
|
+
timer.unref?.();
|
|
799
|
+
this.acks.set(eventId, {
|
|
800
|
+
ackType,
|
|
801
|
+
resolve,
|
|
802
|
+
timer
|
|
803
|
+
});
|
|
804
|
+
});
|
|
805
|
+
}
|
|
806
|
+
resolveAck(clientEventId, acked) {
|
|
807
|
+
if (typeof clientEventId !== "string") return;
|
|
808
|
+
const waiter = this.acks.get(clientEventId);
|
|
809
|
+
if (!waiter) return;
|
|
810
|
+
this.acks.delete(clientEventId);
|
|
811
|
+
clearTimeout(waiter.timer);
|
|
812
|
+
waiter.resolve(acked);
|
|
813
|
+
}
|
|
814
|
+
failPendingAcks() {
|
|
815
|
+
for (const [id, waiter] of this.acks) {
|
|
816
|
+
clearTimeout(waiter.timer);
|
|
817
|
+
waiter.resolve(false);
|
|
818
|
+
this.acks.delete(id);
|
|
819
|
+
}
|
|
820
|
+
}
|
|
821
|
+
sessionConfig() {
|
|
822
|
+
return {
|
|
823
|
+
defaultVoice: this.config.voice,
|
|
824
|
+
delegation: this.config.delegation,
|
|
825
|
+
store: this.config.store,
|
|
826
|
+
extraSessionOptions: this.config.extraSessionOptions,
|
|
827
|
+
historyMaxMessages: this.config.historyMaxMessages,
|
|
828
|
+
historyMaxChars: this.config.historyMaxChars
|
|
829
|
+
};
|
|
830
|
+
}
|
|
831
|
+
isRetriableClose(code) {
|
|
832
|
+
if (this.intentionalClose) return false;
|
|
833
|
+
if (NON_RETRIABLE_CLOSE_CODES.has(code)) return false;
|
|
834
|
+
const reason = this.sessionClosed?.reason;
|
|
835
|
+
if (reason && reason !== "expired" && reason !== "connection_lost") return false;
|
|
836
|
+
return true;
|
|
837
|
+
}
|
|
838
|
+
parseEvent(raw) {
|
|
839
|
+
try {
|
|
840
|
+
return JSON.parse(raw.toString());
|
|
841
|
+
} catch {
|
|
842
|
+
this.logger.warn("unparseable provider frame");
|
|
843
|
+
return null;
|
|
844
|
+
}
|
|
845
|
+
}
|
|
846
|
+
send(payload) {
|
|
847
|
+
if (this.ws?.readyState !== ws.default.OPEN) return;
|
|
848
|
+
try {
|
|
849
|
+
this.ws.send(JSON.stringify(payload));
|
|
850
|
+
} catch (error) {
|
|
851
|
+
this.emit("error", error instanceof Error ? error : new Error(String(error)));
|
|
852
|
+
}
|
|
853
|
+
}
|
|
854
|
+
};
|
|
855
|
+
function numberOr(value, fallback) {
|
|
856
|
+
return typeof value === "number" ? value : fallback;
|
|
857
|
+
}
|
|
858
|
+
/** Split on whitespace so no chunk exceeds the per-append cap. */
|
|
859
|
+
function splitForAppend(content, maxChars = APPEND_MAX_CHARS) {
|
|
860
|
+
const text = content.trim();
|
|
861
|
+
if (text.length <= maxChars) return [text];
|
|
862
|
+
const chunks = [];
|
|
863
|
+
let rest = text;
|
|
864
|
+
while (rest.length > maxChars) {
|
|
865
|
+
let cut = rest.lastIndexOf(" ", maxChars);
|
|
866
|
+
if (cut < maxChars / 2) cut = maxChars;
|
|
867
|
+
chunks.push(rest.slice(0, cut).trim());
|
|
868
|
+
rest = rest.slice(cut).trim();
|
|
869
|
+
}
|
|
870
|
+
if (rest.length) chunks.push(rest);
|
|
871
|
+
return chunks;
|
|
872
|
+
}
|
|
873
|
+
//#endregion
|
|
874
|
+
//#region src/gpt-live.ts
|
|
875
|
+
/** OpenAI GPT-Live provider (full-duplex Live API) — `realtime-voice-agents/gpt-live`. */
|
|
876
|
+
/** Env vars checked (in order) when no explicit apiKey is passed — the same project key as Realtime. */
|
|
877
|
+
const GPT_LIVE_KEY_ENV_VARS = [
|
|
878
|
+
"OPENAI_API_KEY",
|
|
879
|
+
"OPENAI_KEY",
|
|
880
|
+
"OPEN_AI_API_KEY"
|
|
881
|
+
];
|
|
882
|
+
const GPT_LIVE_DEFAULT_MODEL = "gpt-live-1";
|
|
883
|
+
const GPT_LIVE_DEFAULT_VOICE = "marin";
|
|
884
|
+
/** Built-in voices accepted by the Live API (the SDK's enum, Sept 2026). */
|
|
885
|
+
const GPT_LIVE_VOICES = [
|
|
886
|
+
"marin",
|
|
887
|
+
"cedar",
|
|
888
|
+
"alloy",
|
|
889
|
+
"ash",
|
|
890
|
+
"ballad",
|
|
891
|
+
"beacon",
|
|
892
|
+
"bossa",
|
|
893
|
+
"cinder",
|
|
894
|
+
"coral",
|
|
895
|
+
"delta",
|
|
896
|
+
"echo",
|
|
897
|
+
"gleam",
|
|
898
|
+
"meridian",
|
|
899
|
+
"quartz",
|
|
900
|
+
"ripple",
|
|
901
|
+
"sage",
|
|
902
|
+
"shimmer",
|
|
903
|
+
"stone",
|
|
904
|
+
"tempo",
|
|
905
|
+
"verse",
|
|
906
|
+
"vesper",
|
|
907
|
+
"willow"
|
|
908
|
+
];
|
|
909
|
+
/**
|
|
910
|
+
* Create a GPT-Live provider factory for the bridge (`provider: gptLive({...})`).
|
|
911
|
+
*
|
|
912
|
+
* Credentials are resolved per call, when the factory runs (see
|
|
913
|
+
* `openaiRealtime` — a missing key fails that call's connect so a fallback
|
|
914
|
+
* chain can absorb it instead of crashing config construction).
|
|
915
|
+
*/
|
|
916
|
+
function gptLive(options = {}) {
|
|
917
|
+
const config = {
|
|
918
|
+
model: options.model ?? "gpt-live-1",
|
|
919
|
+
voice: options.voice ?? "marin",
|
|
920
|
+
delegation: options.delegation,
|
|
921
|
+
baseUrl: options.baseUrl,
|
|
922
|
+
headers: options.headers,
|
|
923
|
+
store: options.store,
|
|
924
|
+
extraSessionOptions: options.sessionOptions,
|
|
925
|
+
connectTimeoutMs: options.connectTimeoutMs,
|
|
926
|
+
speechGate: options.speechGate,
|
|
927
|
+
transcriptGapMs: options.transcriptGapMs
|
|
928
|
+
};
|
|
929
|
+
return ({ logger }) => {
|
|
930
|
+
const apiKey = require_env.resolveApiKey(options.apiKey, GPT_LIVE_KEY_ENV_VARS);
|
|
931
|
+
if (!apiKey) throw new Error(`gptLive: apiKey missing (pass apiKey or set one of ${GPT_LIVE_KEY_ENV_VARS.join("/")})`);
|
|
932
|
+
return new GptLiveProvider({
|
|
933
|
+
...config,
|
|
934
|
+
apiKey
|
|
935
|
+
}, logger);
|
|
936
|
+
};
|
|
937
|
+
}
|
|
938
|
+
//#endregion
|
|
939
|
+
exports.DEFAULT_GATE_QUIET_MS = DEFAULT_GATE_QUIET_MS;
|
|
940
|
+
exports.DEFAULT_GATE_THRESHOLD_RMS = DEFAULT_GATE_THRESHOLD_RMS;
|
|
941
|
+
exports.GPT_LIVE_AUDIO_FORMAT = GPT_LIVE_AUDIO_FORMAT;
|
|
942
|
+
exports.GPT_LIVE_DEFAULT_BACKEND_MODEL = GPT_LIVE_DEFAULT_BACKEND_MODEL;
|
|
943
|
+
exports.GPT_LIVE_DEFAULT_BASE_URL = GPT_LIVE_DEFAULT_BASE_URL;
|
|
944
|
+
exports.GPT_LIVE_DEFAULT_MODEL = GPT_LIVE_DEFAULT_MODEL;
|
|
945
|
+
exports.GPT_LIVE_DEFAULT_VOICE = GPT_LIVE_DEFAULT_VOICE;
|
|
946
|
+
exports.GPT_LIVE_KEY_ENV_VARS = GPT_LIVE_KEY_ENV_VARS;
|
|
947
|
+
exports.GPT_LIVE_VOICES = GPT_LIVE_VOICES;
|
|
948
|
+
exports.GptLiveProvider = GptLiveProvider;
|
|
949
|
+
exports.SpeechGate = SpeechGate;
|
|
950
|
+
exports.buildDelegation = buildDelegation;
|
|
951
|
+
exports.buildHistoryItems = buildHistoryItems;
|
|
952
|
+
exports.buildSessionStart = buildSessionStart;
|
|
953
|
+
exports.gptLive = gptLive;
|
|
954
|
+
exports.splitForAppend = splitForAppend;
|