realtime-voice-agents 2.5.1 → 2.5.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -1
- package/dist/gpt-live.cjs +78 -7
- package/dist/gpt-live.d.cts +34 -1
- package/dist/gpt-live.d.mts +34 -1
- package/dist/gpt-live.mjs +78 -8
- package/dist/index.cjs +7 -2
- package/dist/index.mjs +7 -2
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -119,8 +119,9 @@ One `SessionOptions` surface configures all four; where a provider can't honor a
|
|
|
119
119
|
- **Tools never pause the voice.** Results are delivered the moment they are ready regardless of `toolResultDelivery`; an interruption does not cancel a running tool, and its result still reaches the backend. Results are relayed in the model's own words — use exact wording only through the voice prompt.
|
|
120
120
|
- **Greetings, nudges and goodbyes** (`greeting.instructions`, `idle.prompts`, `finish_call`) are delivered as `session.commentary.append` — the append that reliably produces speech on demand. Keypad entries and deferred results are `session.thinking.append`; runtime instructions are `session.instructions.append`. Each append is capped at 500 tokens (long texts are split).
|
|
121
121
|
- **Immutable session.** Instructions, voice and audio format cannot change after start, so handoffs and reconnects open a fresh session and seed the attributed transcript through `session.input` (≤ 128 messages) — the anti-loop replay is preserved. Sessions expire after 120 minutes; an expiry reconnects the same way.
|
|
122
|
-
- **Transfers wait for the sentence.** A handoff here is a close-and-reopen, and the backend's transfer lands while the voice is still announcing it — the bridge holds the handoff until that utterance has played out (plus one sentence gap, capped at
|
|
122
|
+
- **Transfers wait for the sentence.** A handoff here is a close-and-reopen, and the backend's transfer lands while the voice is still announcing it — the bridge holds the handoff until that utterance has played out (plus one sentence gap, capped at 8 s), so nothing is cut mid-word and `session.handoffHold` audio covers the reopen. Prompt the voice to *delegate first, announce after*: a transfer or tool the voice announces without delegating never happens.
|
|
123
123
|
- **Deafness feeds silence.** The model's session clock runs on input audio, so `deafness` options replace caller audio with silence instead of dropping frames. `ignoreUserAudioUntilFirstTurnDone` therefore defaults to **off** here — the model handles talk-over itself; set it explicitly to keep the greeting deaf.
|
|
124
|
+
- **Real-time stream, 200 ms of cushion.** The voice arrives at exactly real-time pace, so Twilio's buffer never runs ahead of playout and every delivery hiccup between OpenAI, your server and Twilio would be an audible gap (Realtime generates faster than real time, so it never has this problem). The provider holds the first 200 ms of each utterance — the last idle delta included, so soft onsets are not clipped — then streams through. `gptLive({ playoutLeadMs })` tunes it, `0` disables; the cost is that much latency on each turn's first word.
|
|
124
125
|
- **Billing is per second** of session (plus backend tokens). `session.usage.audioSeconds` carries the running total; backend token usage is summed from `response.completed`. The provider sends `session.close` on teardown and waits for `session.closed`, so a hung-up call never keeps billing.
|
|
125
126
|
|
|
126
127
|
## Provider fallbacks
|
package/dist/gpt-live.cjs
CHANGED
|
@@ -310,6 +310,8 @@ const NON_RETRIABLE_CLOSE_CODES = /* @__PURE__ */ new Set([
|
|
|
310
310
|
const APPEND_MAX_CHARS = 1200;
|
|
311
311
|
const TRANSCRIPT_IDLE_EXTRA_MS = 300;
|
|
312
312
|
const GATE_STALL_EXTRA_MS = 500;
|
|
313
|
+
const DEFAULT_PLAYOUT_LEAD_MS = 200;
|
|
314
|
+
const MULAW_BYTES_PER_MS = 8;
|
|
313
315
|
const ACK_TYPES = {
|
|
314
316
|
"session.instructions.append": "session.instructions.appended",
|
|
315
317
|
"session.thinking.append": "session.thinking.appended",
|
|
@@ -346,6 +348,10 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
|
|
|
346
348
|
expiresAt = null;
|
|
347
349
|
gate;
|
|
348
350
|
gateStallTimer = null;
|
|
351
|
+
/** Deltas of the current utterance held until the playout lead has accumulated. */
|
|
352
|
+
lead = null;
|
|
353
|
+
/** The last idle delta: becomes the utterance's pre-roll when the gate opens. */
|
|
354
|
+
preRoll = null;
|
|
349
355
|
utteranceCounter = 0;
|
|
350
356
|
currentUtteranceId = null;
|
|
351
357
|
lastUtteranceId = null;
|
|
@@ -492,6 +498,8 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
|
|
|
492
498
|
this.inputTranscript.dispose();
|
|
493
499
|
this.outputTranscript.dispose();
|
|
494
500
|
this.clearGateStall();
|
|
501
|
+
this.lead = null;
|
|
502
|
+
this.preRoll = null;
|
|
495
503
|
if (!ws$2 || ws$2.readyState === ws.default.CLOSED) return;
|
|
496
504
|
await new Promise((resolve) => {
|
|
497
505
|
const timer = setTimeout(() => finish(), this.config.closeTimeoutMs ?? 3e3);
|
|
@@ -681,17 +689,74 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
|
|
|
681
689
|
const wasOpen = this.gate.isOpen;
|
|
682
690
|
const events = this.gate.feed(bytes);
|
|
683
691
|
let forwardId = wasOpen ? this.currentUtteranceId : null;
|
|
684
|
-
let
|
|
685
|
-
for (const gateEvent of events)
|
|
692
|
+
let pendingClose = false;
|
|
693
|
+
for (const gateEvent of events) {
|
|
694
|
+
if (gateEvent.type === "close") {
|
|
695
|
+
pendingClose = true;
|
|
696
|
+
continue;
|
|
697
|
+
}
|
|
698
|
+
if (pendingClose) {
|
|
699
|
+
this.endUtterance();
|
|
700
|
+
pendingClose = false;
|
|
701
|
+
}
|
|
686
702
|
this.beginUtterance();
|
|
687
703
|
forwardId = this.currentUtteranceId;
|
|
688
|
-
|
|
689
|
-
|
|
704
|
+
this.armPlayoutLead();
|
|
705
|
+
}
|
|
706
|
+
if (forwardId) this.forwardAudio(delta, bytes.length / MULAW_BYTES_PER_MS, forwardId);
|
|
707
|
+
else this.preRoll = {
|
|
708
|
+
delta,
|
|
709
|
+
ms: bytes.length / MULAW_BYTES_PER_MS
|
|
710
|
+
};
|
|
711
|
+
if (pendingClose) this.endUtterance();
|
|
712
|
+
else if (this.gate.isOpen) this.armGateStall();
|
|
713
|
+
}
|
|
714
|
+
/**
|
|
715
|
+
* The stream is real-time paced, so Twilio's buffer never runs ahead of
|
|
716
|
+
* playout and every delivery hiccup between the model, this server and
|
|
717
|
+
* Twilio is an audible gap (field, Sept 2026: "slightly choppy"). Hold the
|
|
718
|
+
* first `playoutLeadMs` of each utterance — the last idle delta included —
|
|
719
|
+
* then flush and stream through: Twilio keeps that much cushion for the
|
|
720
|
+
* rest of the utterance, at the cost of that much latency on its first word.
|
|
721
|
+
*/
|
|
722
|
+
armPlayoutLead() {
|
|
723
|
+
if (this.playoutLeadMs() <= 0) {
|
|
724
|
+
this.preRoll = null;
|
|
725
|
+
return;
|
|
726
|
+
}
|
|
727
|
+
this.lead = {
|
|
728
|
+
pending: [],
|
|
729
|
+
pendingMs: 0
|
|
730
|
+
};
|
|
731
|
+
if (this.preRoll) {
|
|
732
|
+
this.lead.pending.push(this.preRoll.delta);
|
|
733
|
+
this.lead.pendingMs += this.preRoll.ms;
|
|
734
|
+
this.preRoll = null;
|
|
735
|
+
}
|
|
736
|
+
}
|
|
737
|
+
forwardAudio(delta, ms, responseId) {
|
|
738
|
+
if (!this.lead) {
|
|
739
|
+
this.emit("audio", {
|
|
740
|
+
base64Mulaw: delta,
|
|
741
|
+
responseId
|
|
742
|
+
});
|
|
743
|
+
return;
|
|
744
|
+
}
|
|
745
|
+
this.lead.pending.push(delta);
|
|
746
|
+
this.lead.pendingMs += ms;
|
|
747
|
+
if (this.lead.pendingMs >= this.playoutLeadMs()) this.flushPlayoutLead(responseId);
|
|
748
|
+
}
|
|
749
|
+
flushPlayoutLead(responseId) {
|
|
750
|
+
const lead = this.lead;
|
|
751
|
+
if (!lead) return;
|
|
752
|
+
this.lead = null;
|
|
753
|
+
for (const delta of lead.pending) this.emit("audio", {
|
|
690
754
|
base64Mulaw: delta,
|
|
691
|
-
responseId
|
|
755
|
+
responseId
|
|
692
756
|
});
|
|
693
|
-
|
|
694
|
-
|
|
757
|
+
}
|
|
758
|
+
playoutLeadMs() {
|
|
759
|
+
return this.config.playoutLeadMs ?? 200;
|
|
695
760
|
}
|
|
696
761
|
handleBackendEvent(inner, delegationId) {
|
|
697
762
|
const type = inner.type ?? "";
|
|
@@ -722,6 +787,7 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
|
|
|
722
787
|
});
|
|
723
788
|
}
|
|
724
789
|
beginUtterance() {
|
|
790
|
+
if (this.currentUtteranceId) this.endUtterance();
|
|
725
791
|
this.currentUtteranceId = `live_utt_${++this.utteranceCounter}`;
|
|
726
792
|
this.lastUtteranceId = this.currentUtteranceId;
|
|
727
793
|
this.emit("responseStarted", { responseId: this.currentUtteranceId });
|
|
@@ -729,6 +795,7 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
|
|
|
729
795
|
endUtterance() {
|
|
730
796
|
this.clearGateStall();
|
|
731
797
|
const id = this.currentUtteranceId;
|
|
798
|
+
if (id) this.flushPlayoutLead(id);
|
|
732
799
|
this.currentUtteranceId = null;
|
|
733
800
|
if (id) this.emit("responseDone", { responseId: id });
|
|
734
801
|
}
|
|
@@ -757,6 +824,8 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
|
|
|
757
824
|
resetUtteranceState() {
|
|
758
825
|
this.clearGateStall();
|
|
759
826
|
this.gate.close();
|
|
827
|
+
this.lead = null;
|
|
828
|
+
this.preRoll = null;
|
|
760
829
|
this.currentUtteranceId = null;
|
|
761
830
|
this.lastUtteranceId = null;
|
|
762
831
|
this.seenCalls.clear();
|
|
@@ -923,6 +992,7 @@ function gptLive(options = {}) {
|
|
|
923
992
|
store: options.store,
|
|
924
993
|
extraSessionOptions: options.sessionOptions,
|
|
925
994
|
connectTimeoutMs: options.connectTimeoutMs,
|
|
995
|
+
playoutLeadMs: options.playoutLeadMs,
|
|
926
996
|
speechGate: options.speechGate,
|
|
927
997
|
transcriptGapMs: options.transcriptGapMs
|
|
928
998
|
};
|
|
@@ -938,6 +1008,7 @@ function gptLive(options = {}) {
|
|
|
938
1008
|
//#endregion
|
|
939
1009
|
exports.DEFAULT_GATE_QUIET_MS = DEFAULT_GATE_QUIET_MS;
|
|
940
1010
|
exports.DEFAULT_GATE_THRESHOLD_RMS = DEFAULT_GATE_THRESHOLD_RMS;
|
|
1011
|
+
exports.DEFAULT_PLAYOUT_LEAD_MS = DEFAULT_PLAYOUT_LEAD_MS;
|
|
941
1012
|
exports.GPT_LIVE_AUDIO_FORMAT = GPT_LIVE_AUDIO_FORMAT;
|
|
942
1013
|
exports.GPT_LIVE_DEFAULT_BACKEND_MODEL = GPT_LIVE_DEFAULT_BACKEND_MODEL;
|
|
943
1014
|
exports.GPT_LIVE_DEFAULT_BASE_URL = GPT_LIVE_DEFAULT_BASE_URL;
|
package/dist/gpt-live.d.cts
CHANGED
|
@@ -126,6 +126,14 @@ interface GptLiveProviderConfig {
|
|
|
126
126
|
connectTimeoutMs?: number;
|
|
127
127
|
/** How long `close()` waits for `session.closed` (final usage) before dropping the socket. Default 3000. */
|
|
128
128
|
closeTimeoutMs?: number;
|
|
129
|
+
/**
|
|
130
|
+
* Audio held back at the start of each utterance before forwarding begins,
|
|
131
|
+
* so Twilio keeps that much cushion against delivery jitter (the stream is
|
|
132
|
+
* real-time paced — without it any hiccup is an audible gap). The last idle
|
|
133
|
+
* delta before the onset is included, so soft onsets are not clipped.
|
|
134
|
+
* Default 200; 0 forwards every delta as it arrives.
|
|
135
|
+
*/
|
|
136
|
+
playoutLeadMs?: number;
|
|
129
137
|
/** Speech gate tuning (utterance boundaries synthesized from the audio). */
|
|
130
138
|
speechGate?: SpeechGateOptions;
|
|
131
139
|
/** Session-timeline gap that splits transcript fragments into turns. Default 800. */
|
|
@@ -137,6 +145,7 @@ interface GptLiveProviderConfig {
|
|
|
137
145
|
ackTimeoutMs?: number;
|
|
138
146
|
providerName?: string;
|
|
139
147
|
}
|
|
148
|
+
declare const DEFAULT_PLAYOUT_LEAD_MS = 200;
|
|
140
149
|
declare class GptLiveProvider extends BaseRealtimeProvider {
|
|
141
150
|
readonly name: string;
|
|
142
151
|
readonly capabilities: ProviderCapabilities;
|
|
@@ -152,6 +161,10 @@ declare class GptLiveProvider extends BaseRealtimeProvider {
|
|
|
152
161
|
expiresAt: number | null;
|
|
153
162
|
private readonly gate;
|
|
154
163
|
private gateStallTimer;
|
|
164
|
+
/** Deltas of the current utterance held until the playout lead has accumulated. */
|
|
165
|
+
private lead;
|
|
166
|
+
/** The last idle delta: becomes the utterance's pre-roll when the gate opens. */
|
|
167
|
+
private preRoll;
|
|
155
168
|
private utteranceCounter;
|
|
156
169
|
private currentUtteranceId;
|
|
157
170
|
private lastUtteranceId;
|
|
@@ -192,6 +205,18 @@ declare class GptLiveProvider extends BaseRealtimeProvider {
|
|
|
192
205
|
updateSession(patch: Partial<ProviderSessionInit>, options?: SessionUpdateOptions): Promise<boolean>;
|
|
193
206
|
private handleEvent;
|
|
194
207
|
private handleOutputAudio;
|
|
208
|
+
/**
|
|
209
|
+
* The stream is real-time paced, so Twilio's buffer never runs ahead of
|
|
210
|
+
* playout and every delivery hiccup between the model, this server and
|
|
211
|
+
* Twilio is an audible gap (field, Sept 2026: "slightly choppy"). Hold the
|
|
212
|
+
* first `playoutLeadMs` of each utterance — the last idle delta included —
|
|
213
|
+
* then flush and stream through: Twilio keeps that much cushion for the
|
|
214
|
+
* rest of the utterance, at the cost of that much latency on its first word.
|
|
215
|
+
*/
|
|
216
|
+
private armPlayoutLead;
|
|
217
|
+
private forwardAudio;
|
|
218
|
+
private flushPlayoutLead;
|
|
219
|
+
private playoutLeadMs;
|
|
195
220
|
private handleBackendEvent;
|
|
196
221
|
private beginUtterance;
|
|
197
222
|
private endUtterance;
|
|
@@ -244,6 +269,14 @@ interface GptLiveOptions {
|
|
|
244
269
|
/** Provider-native `session.start` fields, deep-merged last. The schema is strict: an unknown field rejects the session. */
|
|
245
270
|
sessionOptions?: Record<string, unknown>;
|
|
246
271
|
connectTimeoutMs?: number;
|
|
272
|
+
/**
|
|
273
|
+
* Cushion held at the start of each utterance before audio is forwarded to
|
|
274
|
+
* Twilio. The model streams at exactly real-time pace, so without it any
|
|
275
|
+
* delivery hiccup is an audible gap; with it Twilio stays that far ahead of
|
|
276
|
+
* playout. Costs the same amount of latency on each turn's first word.
|
|
277
|
+
* Default 200; 0 forwards every delta as it arrives.
|
|
278
|
+
*/
|
|
279
|
+
playoutLeadMs?: number;
|
|
247
280
|
/** Speech gate tuning — utterance boundaries synthesized from the continuous output stream. */
|
|
248
281
|
speechGate?: SpeechGateOptions;
|
|
249
282
|
/** Session-timeline gap that splits transcript fragments into turns. Default 800. */
|
|
@@ -258,4 +291,4 @@ interface GptLiveOptions {
|
|
|
258
291
|
*/
|
|
259
292
|
declare function gptLive(options?: GptLiveOptions): ProviderFactory;
|
|
260
293
|
//#endregion
|
|
261
|
-
export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, type GptLiveDelegationOptions, GptLiveOptions, GptLiveProvider, type GptLiveProviderConfig, type GptLiveSessionConfig, GptLiveVoice, SpeechGate, type SpeechGateOptions, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
|
|
294
|
+
export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, DEFAULT_PLAYOUT_LEAD_MS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, type GptLiveDelegationOptions, GptLiveOptions, GptLiveProvider, type GptLiveProviderConfig, type GptLiveSessionConfig, GptLiveVoice, SpeechGate, type SpeechGateOptions, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
|
package/dist/gpt-live.d.mts
CHANGED
|
@@ -126,6 +126,14 @@ interface GptLiveProviderConfig {
|
|
|
126
126
|
connectTimeoutMs?: number;
|
|
127
127
|
/** How long `close()` waits for `session.closed` (final usage) before dropping the socket. Default 3000. */
|
|
128
128
|
closeTimeoutMs?: number;
|
|
129
|
+
/**
|
|
130
|
+
* Audio held back at the start of each utterance before forwarding begins,
|
|
131
|
+
* so Twilio keeps that much cushion against delivery jitter (the stream is
|
|
132
|
+
* real-time paced — without it any hiccup is an audible gap). The last idle
|
|
133
|
+
* delta before the onset is included, so soft onsets are not clipped.
|
|
134
|
+
* Default 200; 0 forwards every delta as it arrives.
|
|
135
|
+
*/
|
|
136
|
+
playoutLeadMs?: number;
|
|
129
137
|
/** Speech gate tuning (utterance boundaries synthesized from the audio). */
|
|
130
138
|
speechGate?: SpeechGateOptions;
|
|
131
139
|
/** Session-timeline gap that splits transcript fragments into turns. Default 800. */
|
|
@@ -137,6 +145,7 @@ interface GptLiveProviderConfig {
|
|
|
137
145
|
ackTimeoutMs?: number;
|
|
138
146
|
providerName?: string;
|
|
139
147
|
}
|
|
148
|
+
declare const DEFAULT_PLAYOUT_LEAD_MS = 200;
|
|
140
149
|
declare class GptLiveProvider extends BaseRealtimeProvider {
|
|
141
150
|
readonly name: string;
|
|
142
151
|
readonly capabilities: ProviderCapabilities;
|
|
@@ -152,6 +161,10 @@ declare class GptLiveProvider extends BaseRealtimeProvider {
|
|
|
152
161
|
expiresAt: number | null;
|
|
153
162
|
private readonly gate;
|
|
154
163
|
private gateStallTimer;
|
|
164
|
+
/** Deltas of the current utterance held until the playout lead has accumulated. */
|
|
165
|
+
private lead;
|
|
166
|
+
/** The last idle delta: becomes the utterance's pre-roll when the gate opens. */
|
|
167
|
+
private preRoll;
|
|
155
168
|
private utteranceCounter;
|
|
156
169
|
private currentUtteranceId;
|
|
157
170
|
private lastUtteranceId;
|
|
@@ -192,6 +205,18 @@ declare class GptLiveProvider extends BaseRealtimeProvider {
|
|
|
192
205
|
updateSession(patch: Partial<ProviderSessionInit>, options?: SessionUpdateOptions): Promise<boolean>;
|
|
193
206
|
private handleEvent;
|
|
194
207
|
private handleOutputAudio;
|
|
208
|
+
/**
|
|
209
|
+
* The stream is real-time paced, so Twilio's buffer never runs ahead of
|
|
210
|
+
* playout and every delivery hiccup between the model, this server and
|
|
211
|
+
* Twilio is an audible gap (field, Sept 2026: "slightly choppy"). Hold the
|
|
212
|
+
* first `playoutLeadMs` of each utterance — the last idle delta included —
|
|
213
|
+
* then flush and stream through: Twilio keeps that much cushion for the
|
|
214
|
+
* rest of the utterance, at the cost of that much latency on its first word.
|
|
215
|
+
*/
|
|
216
|
+
private armPlayoutLead;
|
|
217
|
+
private forwardAudio;
|
|
218
|
+
private flushPlayoutLead;
|
|
219
|
+
private playoutLeadMs;
|
|
195
220
|
private handleBackendEvent;
|
|
196
221
|
private beginUtterance;
|
|
197
222
|
private endUtterance;
|
|
@@ -244,6 +269,14 @@ interface GptLiveOptions {
|
|
|
244
269
|
/** Provider-native `session.start` fields, deep-merged last. The schema is strict: an unknown field rejects the session. */
|
|
245
270
|
sessionOptions?: Record<string, unknown>;
|
|
246
271
|
connectTimeoutMs?: number;
|
|
272
|
+
/**
|
|
273
|
+
* Cushion held at the start of each utterance before audio is forwarded to
|
|
274
|
+
* Twilio. The model streams at exactly real-time pace, so without it any
|
|
275
|
+
* delivery hiccup is an audible gap; with it Twilio stays that far ahead of
|
|
276
|
+
* playout. Costs the same amount of latency on each turn's first word.
|
|
277
|
+
* Default 200; 0 forwards every delta as it arrives.
|
|
278
|
+
*/
|
|
279
|
+
playoutLeadMs?: number;
|
|
247
280
|
/** Speech gate tuning — utterance boundaries synthesized from the continuous output stream. */
|
|
248
281
|
speechGate?: SpeechGateOptions;
|
|
249
282
|
/** Session-timeline gap that splits transcript fragments into turns. Default 800. */
|
|
@@ -258,4 +291,4 @@ interface GptLiveOptions {
|
|
|
258
291
|
*/
|
|
259
292
|
declare function gptLive(options?: GptLiveOptions): ProviderFactory;
|
|
260
293
|
//#endregion
|
|
261
|
-
export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, type GptLiveDelegationOptions, GptLiveOptions, GptLiveProvider, type GptLiveProviderConfig, type GptLiveSessionConfig, GptLiveVoice, SpeechGate, type SpeechGateOptions, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
|
|
294
|
+
export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, DEFAULT_PLAYOUT_LEAD_MS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, type GptLiveDelegationOptions, GptLiveOptions, GptLiveProvider, type GptLiveProviderConfig, type GptLiveSessionConfig, GptLiveVoice, SpeechGate, type SpeechGateOptions, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
|
package/dist/gpt-live.mjs
CHANGED
|
@@ -307,6 +307,8 @@ const NON_RETRIABLE_CLOSE_CODES = /* @__PURE__ */ new Set([
|
|
|
307
307
|
const APPEND_MAX_CHARS = 1200;
|
|
308
308
|
const TRANSCRIPT_IDLE_EXTRA_MS = 300;
|
|
309
309
|
const GATE_STALL_EXTRA_MS = 500;
|
|
310
|
+
const DEFAULT_PLAYOUT_LEAD_MS = 200;
|
|
311
|
+
const MULAW_BYTES_PER_MS = 8;
|
|
310
312
|
const ACK_TYPES = {
|
|
311
313
|
"session.instructions.append": "session.instructions.appended",
|
|
312
314
|
"session.thinking.append": "session.thinking.appended",
|
|
@@ -343,6 +345,10 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
|
|
|
343
345
|
expiresAt = null;
|
|
344
346
|
gate;
|
|
345
347
|
gateStallTimer = null;
|
|
348
|
+
/** Deltas of the current utterance held until the playout lead has accumulated. */
|
|
349
|
+
lead = null;
|
|
350
|
+
/** The last idle delta: becomes the utterance's pre-roll when the gate opens. */
|
|
351
|
+
preRoll = null;
|
|
346
352
|
utteranceCounter = 0;
|
|
347
353
|
currentUtteranceId = null;
|
|
348
354
|
lastUtteranceId = null;
|
|
@@ -489,6 +495,8 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
|
|
|
489
495
|
this.inputTranscript.dispose();
|
|
490
496
|
this.outputTranscript.dispose();
|
|
491
497
|
this.clearGateStall();
|
|
498
|
+
this.lead = null;
|
|
499
|
+
this.preRoll = null;
|
|
492
500
|
if (!ws || ws.readyState === WebSocket$1.CLOSED) return;
|
|
493
501
|
await new Promise((resolve) => {
|
|
494
502
|
const timer = setTimeout(() => finish(), this.config.closeTimeoutMs ?? 3e3);
|
|
@@ -678,17 +686,74 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
|
|
|
678
686
|
const wasOpen = this.gate.isOpen;
|
|
679
687
|
const events = this.gate.feed(bytes);
|
|
680
688
|
let forwardId = wasOpen ? this.currentUtteranceId : null;
|
|
681
|
-
let
|
|
682
|
-
for (const gateEvent of events)
|
|
689
|
+
let pendingClose = false;
|
|
690
|
+
for (const gateEvent of events) {
|
|
691
|
+
if (gateEvent.type === "close") {
|
|
692
|
+
pendingClose = true;
|
|
693
|
+
continue;
|
|
694
|
+
}
|
|
695
|
+
if (pendingClose) {
|
|
696
|
+
this.endUtterance();
|
|
697
|
+
pendingClose = false;
|
|
698
|
+
}
|
|
683
699
|
this.beginUtterance();
|
|
684
700
|
forwardId = this.currentUtteranceId;
|
|
685
|
-
|
|
686
|
-
|
|
701
|
+
this.armPlayoutLead();
|
|
702
|
+
}
|
|
703
|
+
if (forwardId) this.forwardAudio(delta, bytes.length / MULAW_BYTES_PER_MS, forwardId);
|
|
704
|
+
else this.preRoll = {
|
|
705
|
+
delta,
|
|
706
|
+
ms: bytes.length / MULAW_BYTES_PER_MS
|
|
707
|
+
};
|
|
708
|
+
if (pendingClose) this.endUtterance();
|
|
709
|
+
else if (this.gate.isOpen) this.armGateStall();
|
|
710
|
+
}
|
|
711
|
+
/**
|
|
712
|
+
* The stream is real-time paced, so Twilio's buffer never runs ahead of
|
|
713
|
+
* playout and every delivery hiccup between the model, this server and
|
|
714
|
+
* Twilio is an audible gap (field, Sept 2026: "slightly choppy"). Hold the
|
|
715
|
+
* first `playoutLeadMs` of each utterance — the last idle delta included —
|
|
716
|
+
* then flush and stream through: Twilio keeps that much cushion for the
|
|
717
|
+
* rest of the utterance, at the cost of that much latency on its first word.
|
|
718
|
+
*/
|
|
719
|
+
armPlayoutLead() {
|
|
720
|
+
if (this.playoutLeadMs() <= 0) {
|
|
721
|
+
this.preRoll = null;
|
|
722
|
+
return;
|
|
723
|
+
}
|
|
724
|
+
this.lead = {
|
|
725
|
+
pending: [],
|
|
726
|
+
pendingMs: 0
|
|
727
|
+
};
|
|
728
|
+
if (this.preRoll) {
|
|
729
|
+
this.lead.pending.push(this.preRoll.delta);
|
|
730
|
+
this.lead.pendingMs += this.preRoll.ms;
|
|
731
|
+
this.preRoll = null;
|
|
732
|
+
}
|
|
733
|
+
}
|
|
734
|
+
forwardAudio(delta, ms, responseId) {
|
|
735
|
+
if (!this.lead) {
|
|
736
|
+
this.emit("audio", {
|
|
737
|
+
base64Mulaw: delta,
|
|
738
|
+
responseId
|
|
739
|
+
});
|
|
740
|
+
return;
|
|
741
|
+
}
|
|
742
|
+
this.lead.pending.push(delta);
|
|
743
|
+
this.lead.pendingMs += ms;
|
|
744
|
+
if (this.lead.pendingMs >= this.playoutLeadMs()) this.flushPlayoutLead(responseId);
|
|
745
|
+
}
|
|
746
|
+
flushPlayoutLead(responseId) {
|
|
747
|
+
const lead = this.lead;
|
|
748
|
+
if (!lead) return;
|
|
749
|
+
this.lead = null;
|
|
750
|
+
for (const delta of lead.pending) this.emit("audio", {
|
|
687
751
|
base64Mulaw: delta,
|
|
688
|
-
responseId
|
|
752
|
+
responseId
|
|
689
753
|
});
|
|
690
|
-
|
|
691
|
-
|
|
754
|
+
}
|
|
755
|
+
playoutLeadMs() {
|
|
756
|
+
return this.config.playoutLeadMs ?? 200;
|
|
692
757
|
}
|
|
693
758
|
handleBackendEvent(inner, delegationId) {
|
|
694
759
|
const type = inner.type ?? "";
|
|
@@ -719,6 +784,7 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
|
|
|
719
784
|
});
|
|
720
785
|
}
|
|
721
786
|
beginUtterance() {
|
|
787
|
+
if (this.currentUtteranceId) this.endUtterance();
|
|
722
788
|
this.currentUtteranceId = `live_utt_${++this.utteranceCounter}`;
|
|
723
789
|
this.lastUtteranceId = this.currentUtteranceId;
|
|
724
790
|
this.emit("responseStarted", { responseId: this.currentUtteranceId });
|
|
@@ -726,6 +792,7 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
|
|
|
726
792
|
endUtterance() {
|
|
727
793
|
this.clearGateStall();
|
|
728
794
|
const id = this.currentUtteranceId;
|
|
795
|
+
if (id) this.flushPlayoutLead(id);
|
|
729
796
|
this.currentUtteranceId = null;
|
|
730
797
|
if (id) this.emit("responseDone", { responseId: id });
|
|
731
798
|
}
|
|
@@ -754,6 +821,8 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
|
|
|
754
821
|
resetUtteranceState() {
|
|
755
822
|
this.clearGateStall();
|
|
756
823
|
this.gate.close();
|
|
824
|
+
this.lead = null;
|
|
825
|
+
this.preRoll = null;
|
|
757
826
|
this.currentUtteranceId = null;
|
|
758
827
|
this.lastUtteranceId = null;
|
|
759
828
|
this.seenCalls.clear();
|
|
@@ -920,6 +989,7 @@ function gptLive(options = {}) {
|
|
|
920
989
|
store: options.store,
|
|
921
990
|
extraSessionOptions: options.sessionOptions,
|
|
922
991
|
connectTimeoutMs: options.connectTimeoutMs,
|
|
992
|
+
playoutLeadMs: options.playoutLeadMs,
|
|
923
993
|
speechGate: options.speechGate,
|
|
924
994
|
transcriptGapMs: options.transcriptGapMs
|
|
925
995
|
};
|
|
@@ -933,4 +1003,4 @@ function gptLive(options = {}) {
|
|
|
933
1003
|
};
|
|
934
1004
|
}
|
|
935
1005
|
//#endregion
|
|
936
|
-
export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, GptLiveProvider, SpeechGate, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
|
|
1006
|
+
export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, DEFAULT_PLAYOUT_LEAD_MS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, GptLiveProvider, SpeechGate, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
|
package/dist/index.cjs
CHANGED
|
@@ -2631,8 +2631,13 @@ const PREGREETING_MARK = "pre:greeting";
|
|
|
2631
2631
|
/** Longer than the speech gate's quiet window plus the mark round trip (0.8 s + ~0.3 s). */
|
|
2632
2632
|
/** Model-owned turn-taking: the gate can split one sentence at a pause — wait one gap for the next. */
|
|
2633
2633
|
const SENTENCE_GRACE_MS = 1500;
|
|
2634
|
-
/**
|
|
2635
|
-
|
|
2634
|
+
/**
|
|
2635
|
+
* A deferred handoff runs no later than this after the transfer landed, even
|
|
2636
|
+
* mid-utterance. One announce + a stray utterance + the sentence grace lands
|
|
2637
|
+
* right at 5 s in the field (Sept 2026); the cap is for a voice that never
|
|
2638
|
+
* stops, and the caller hears the agent meanwhile, so it errs long.
|
|
2639
|
+
*/
|
|
2640
|
+
const HANDOFF_DEFER_MAX_MS = 8e3;
|
|
2636
2641
|
/** 400ms per frame — matches production burst-write implementations. */
|
|
2637
2642
|
const PREGREETING_CHUNK_BYTES = 3200;
|
|
2638
2643
|
function toError(value) {
|
package/dist/index.mjs
CHANGED
|
@@ -2627,8 +2627,13 @@ const PREGREETING_MARK = "pre:greeting";
|
|
|
2627
2627
|
/** Longer than the speech gate's quiet window plus the mark round trip (0.8 s + ~0.3 s). */
|
|
2628
2628
|
/** Model-owned turn-taking: the gate can split one sentence at a pause — wait one gap for the next. */
|
|
2629
2629
|
const SENTENCE_GRACE_MS = 1500;
|
|
2630
|
-
/**
|
|
2631
|
-
|
|
2630
|
+
/**
|
|
2631
|
+
* A deferred handoff runs no later than this after the transfer landed, even
|
|
2632
|
+
* mid-utterance. One announce + a stray utterance + the sentence grace lands
|
|
2633
|
+
* right at 5 s in the field (Sept 2026); the cap is for a voice that never
|
|
2634
|
+
* stops, and the caller hears the agent meanwhile, so it errs long.
|
|
2635
|
+
*/
|
|
2636
|
+
const HANDOFF_DEFER_MAX_MS = 8e3;
|
|
2632
2637
|
/** 400ms per frame — matches production burst-write implementations. */
|
|
2633
2638
|
const PREGREETING_CHUNK_BYTES = 3200;
|
|
2634
2639
|
function toError(value) {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "realtime-voice-agents",
|
|
3
|
-
"version": "2.5.
|
|
3
|
+
"version": "2.5.3",
|
|
4
4
|
"description": "Provider-agnostic bridge between Twilio Media Streams and realtime speech-to-speech AI APIs (OpenAI Realtime, OpenAI GPT-Live full-duplex, xAI Grok Voice, Gemini Live). Multi-agent handoffs, Zod tools with execution strategies, mark-based playback tracking, interruption guards, and hold audio — for Node.js voice agents over the phone.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"twilio",
|