realtime-voice-agents 2.5.0 → 2.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -2
- package/dist/gpt-live.cjs +66 -4
- package/dist/gpt-live.d.cts +34 -1
- package/dist/gpt-live.d.mts +34 -1
- package/dist/gpt-live.mjs +66 -5
- package/dist/index.cjs +126 -4
- package/dist/index.d.cts +50 -5
- package/dist/index.d.mts +50 -5
- package/dist/index.mjs +126 -4
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -119,7 +119,9 @@ One `SessionOptions` surface configures all four; where a provider can't honor a
|
|
|
119
119
|
- **Tools never pause the voice.** Results are delivered the moment they are ready regardless of `toolResultDelivery`; an interruption does not cancel a running tool, and its result still reaches the backend. Results are relayed in the model's own words — use exact wording only through the voice prompt.
|
|
120
120
|
- **Greetings, nudges and goodbyes** (`greeting.instructions`, `idle.prompts`, `finish_call`) are delivered as `session.commentary.append` — the append that reliably produces speech on demand. Keypad entries and deferred results are `session.thinking.append`; runtime instructions are `session.instructions.append`. Each append is capped at 500 tokens (long texts are split).
|
|
121
121
|
- **Immutable session.** Instructions, voice and audio format cannot change after start, so handoffs and reconnects open a fresh session and seed the attributed transcript through `session.input` (≤ 128 messages) — the anti-loop replay is preserved. Sessions expire after 120 minutes; an expiry reconnects the same way.
|
|
122
|
-
- **
|
|
122
|
+
- **Transfers wait for the sentence.** A handoff here is a close-and-reopen, and the backend's transfer lands while the voice is still announcing it — the bridge holds the handoff until that utterance has played out (plus one sentence gap, capped at 5 s), so nothing is cut mid-word and `session.handoffHold` audio covers the reopen. Prompt the voice to *delegate first, announce after*: a transfer or tool the voice announces without delegating never happens.
|
|
123
|
+
- **Deafness feeds silence.** The model's session clock runs on input audio, so `deafness` options replace caller audio with silence instead of dropping frames. `ignoreUserAudioUntilFirstTurnDone` therefore defaults to **off** here — the model handles talk-over itself; set it explicitly to keep the greeting deaf.
|
|
124
|
+
- **Real-time stream, 200 ms of cushion.** The voice arrives at exactly real-time pace, so Twilio's buffer never runs ahead of playout and every delivery hiccup between OpenAI, your server and Twilio would be an audible gap (Realtime generates faster than real time, so it never has this problem). The provider holds the first 200 ms of each utterance — the last idle delta included, so soft onsets are not clipped — then streams through. `gptLive({ playoutLeadMs })` tunes it, `0` disables; the cost is that much latency on each turn's first word.
|
|
123
125
|
- **Billing is per second** of session (plus backend tokens). `session.usage.audioSeconds` carries the running total; backend token usage is summed from `response.completed`. The provider sends `session.close` on teardown and waits for `session.closed`, so a hung-up call never keeps billing.
|
|
124
126
|
|
|
125
127
|
## Provider fallbacks
|
|
@@ -341,7 +343,7 @@ session: {
|
|
|
341
343
|
greeting: { mode: 'agent-initiates' }, // 'user-initiates' to wait
|
|
342
344
|
interruptions: { enabled: true },
|
|
343
345
|
deafness: {
|
|
344
|
-
ignoreUserAudioUntilFirstTurnDone:
|
|
346
|
+
ignoreUserAudioUntilFirstTurnDone: undefined, // auto: true, except false with greeting.mode 'user-initiates' and on full-duplex providers (GPT-Live)
|
|
345
347
|
muteDuringToolExecution: true,
|
|
346
348
|
muteWhileAgentSpeaking: false, // half-duplex: deaf while agent audio plays (caller speech is lost, not queued)
|
|
347
349
|
},
|
|
@@ -355,6 +357,7 @@ session: {
|
|
|
355
357
|
toolResultDelivery: 'afterPlayback', // or 'immediate'
|
|
356
358
|
toolBackgroundAudio: undefined, // default hold audio for tools
|
|
357
359
|
handoffVoicePolicy: 'keep', // or 'reconnect' to switch voices
|
|
360
|
+
handoffHold: undefined, // { spec: 'ringing', ... }: hold audio over a reconnect-style handoff (Gemini, GPT-Live)
|
|
358
361
|
context: {}, // seed session KV for tools/instructions
|
|
359
362
|
}
|
|
360
363
|
```
|
package/dist/gpt-live.cjs
CHANGED
|
@@ -310,6 +310,8 @@ const NON_RETRIABLE_CLOSE_CODES = /* @__PURE__ */ new Set([
|
|
|
310
310
|
const APPEND_MAX_CHARS = 1200;
|
|
311
311
|
const TRANSCRIPT_IDLE_EXTRA_MS = 300;
|
|
312
312
|
const GATE_STALL_EXTRA_MS = 500;
|
|
313
|
+
const DEFAULT_PLAYOUT_LEAD_MS = 200;
|
|
314
|
+
const MULAW_BYTES_PER_MS = 8;
|
|
313
315
|
const ACK_TYPES = {
|
|
314
316
|
"session.instructions.append": "session.instructions.appended",
|
|
315
317
|
"session.thinking.append": "session.thinking.appended",
|
|
@@ -346,6 +348,10 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
|
|
|
346
348
|
expiresAt = null;
|
|
347
349
|
gate;
|
|
348
350
|
gateStallTimer = null;
|
|
351
|
+
/** Deltas of the current utterance held until the playout lead has accumulated. */
|
|
352
|
+
lead = null;
|
|
353
|
+
/** The last idle delta: becomes the utterance's pre-roll when the gate opens. */
|
|
354
|
+
preRoll = null;
|
|
349
355
|
utteranceCounter = 0;
|
|
350
356
|
currentUtteranceId = null;
|
|
351
357
|
lastUtteranceId = null;
|
|
@@ -492,6 +498,8 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
|
|
|
492
498
|
this.inputTranscript.dispose();
|
|
493
499
|
this.outputTranscript.dispose();
|
|
494
500
|
this.clearGateStall();
|
|
501
|
+
this.lead = null;
|
|
502
|
+
this.preRoll = null;
|
|
495
503
|
if (!ws$2 || ws$2.readyState === ws.default.CLOSED) return;
|
|
496
504
|
await new Promise((resolve) => {
|
|
497
505
|
const timer = setTimeout(() => finish(), this.config.closeTimeoutMs ?? 3e3);
|
|
@@ -685,14 +693,63 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
|
|
|
685
693
|
for (const gateEvent of events) if (gateEvent.type === "open") {
|
|
686
694
|
this.beginUtterance();
|
|
687
695
|
forwardId = this.currentUtteranceId;
|
|
696
|
+
this.armPlayoutLead();
|
|
688
697
|
} else close = gateEvent;
|
|
689
|
-
if (forwardId) this.
|
|
690
|
-
|
|
691
|
-
|
|
692
|
-
|
|
698
|
+
if (forwardId) this.forwardAudio(delta, bytes.length / MULAW_BYTES_PER_MS, forwardId);
|
|
699
|
+
else this.preRoll = {
|
|
700
|
+
delta,
|
|
701
|
+
ms: bytes.length / MULAW_BYTES_PER_MS
|
|
702
|
+
};
|
|
693
703
|
if (close) this.endUtterance();
|
|
694
704
|
else if (this.gate.isOpen) this.armGateStall();
|
|
695
705
|
}
|
|
706
|
+
/**
|
|
707
|
+
* The stream is real-time paced, so Twilio's buffer never runs ahead of
|
|
708
|
+
* playout and every delivery hiccup between the model, this server and
|
|
709
|
+
* Twilio is an audible gap (field, Sept 2026: "slightly choppy"). Hold the
|
|
710
|
+
* first `playoutLeadMs` of each utterance — the last idle delta included —
|
|
711
|
+
* then flush and stream through: Twilio keeps that much cushion for the
|
|
712
|
+
* rest of the utterance, at the cost of that much latency on its first word.
|
|
713
|
+
*/
|
|
714
|
+
armPlayoutLead() {
|
|
715
|
+
if (this.playoutLeadMs() <= 0) {
|
|
716
|
+
this.preRoll = null;
|
|
717
|
+
return;
|
|
718
|
+
}
|
|
719
|
+
this.lead = {
|
|
720
|
+
pending: [],
|
|
721
|
+
pendingMs: 0
|
|
722
|
+
};
|
|
723
|
+
if (this.preRoll) {
|
|
724
|
+
this.lead.pending.push(this.preRoll.delta);
|
|
725
|
+
this.lead.pendingMs += this.preRoll.ms;
|
|
726
|
+
this.preRoll = null;
|
|
727
|
+
}
|
|
728
|
+
}
|
|
729
|
+
forwardAudio(delta, ms, responseId) {
|
|
730
|
+
if (!this.lead) {
|
|
731
|
+
this.emit("audio", {
|
|
732
|
+
base64Mulaw: delta,
|
|
733
|
+
responseId
|
|
734
|
+
});
|
|
735
|
+
return;
|
|
736
|
+
}
|
|
737
|
+
this.lead.pending.push(delta);
|
|
738
|
+
this.lead.pendingMs += ms;
|
|
739
|
+
if (this.lead.pendingMs >= this.playoutLeadMs()) this.flushPlayoutLead(responseId);
|
|
740
|
+
}
|
|
741
|
+
flushPlayoutLead(responseId) {
|
|
742
|
+
const lead = this.lead;
|
|
743
|
+
if (!lead) return;
|
|
744
|
+
this.lead = null;
|
|
745
|
+
for (const delta of lead.pending) this.emit("audio", {
|
|
746
|
+
base64Mulaw: delta,
|
|
747
|
+
responseId
|
|
748
|
+
});
|
|
749
|
+
}
|
|
750
|
+
playoutLeadMs() {
|
|
751
|
+
return this.config.playoutLeadMs ?? 200;
|
|
752
|
+
}
|
|
696
753
|
handleBackendEvent(inner, delegationId) {
|
|
697
754
|
const type = inner.type ?? "";
|
|
698
755
|
if (type === "response.output_item.done" && inner.item?.type === "function_call") {
|
|
@@ -729,6 +786,7 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
|
|
|
729
786
|
endUtterance() {
|
|
730
787
|
this.clearGateStall();
|
|
731
788
|
const id = this.currentUtteranceId;
|
|
789
|
+
if (id) this.flushPlayoutLead(id);
|
|
732
790
|
this.currentUtteranceId = null;
|
|
733
791
|
if (id) this.emit("responseDone", { responseId: id });
|
|
734
792
|
}
|
|
@@ -757,6 +815,8 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
|
|
|
757
815
|
resetUtteranceState() {
|
|
758
816
|
this.clearGateStall();
|
|
759
817
|
this.gate.close();
|
|
818
|
+
this.lead = null;
|
|
819
|
+
this.preRoll = null;
|
|
760
820
|
this.currentUtteranceId = null;
|
|
761
821
|
this.lastUtteranceId = null;
|
|
762
822
|
this.seenCalls.clear();
|
|
@@ -923,6 +983,7 @@ function gptLive(options = {}) {
|
|
|
923
983
|
store: options.store,
|
|
924
984
|
extraSessionOptions: options.sessionOptions,
|
|
925
985
|
connectTimeoutMs: options.connectTimeoutMs,
|
|
986
|
+
playoutLeadMs: options.playoutLeadMs,
|
|
926
987
|
speechGate: options.speechGate,
|
|
927
988
|
transcriptGapMs: options.transcriptGapMs
|
|
928
989
|
};
|
|
@@ -938,6 +999,7 @@ function gptLive(options = {}) {
|
|
|
938
999
|
//#endregion
|
|
939
1000
|
exports.DEFAULT_GATE_QUIET_MS = DEFAULT_GATE_QUIET_MS;
|
|
940
1001
|
exports.DEFAULT_GATE_THRESHOLD_RMS = DEFAULT_GATE_THRESHOLD_RMS;
|
|
1002
|
+
exports.DEFAULT_PLAYOUT_LEAD_MS = DEFAULT_PLAYOUT_LEAD_MS;
|
|
941
1003
|
exports.GPT_LIVE_AUDIO_FORMAT = GPT_LIVE_AUDIO_FORMAT;
|
|
942
1004
|
exports.GPT_LIVE_DEFAULT_BACKEND_MODEL = GPT_LIVE_DEFAULT_BACKEND_MODEL;
|
|
943
1005
|
exports.GPT_LIVE_DEFAULT_BASE_URL = GPT_LIVE_DEFAULT_BASE_URL;
|
package/dist/gpt-live.d.cts
CHANGED
|
@@ -126,6 +126,14 @@ interface GptLiveProviderConfig {
|
|
|
126
126
|
connectTimeoutMs?: number;
|
|
127
127
|
/** How long `close()` waits for `session.closed` (final usage) before dropping the socket. Default 3000. */
|
|
128
128
|
closeTimeoutMs?: number;
|
|
129
|
+
/**
|
|
130
|
+
* Audio held back at the start of each utterance before forwarding begins,
|
|
131
|
+
* so Twilio keeps that much cushion against delivery jitter (the stream is
|
|
132
|
+
* real-time paced — without it any hiccup is an audible gap). The last idle
|
|
133
|
+
* delta before the onset is included, so soft onsets are not clipped.
|
|
134
|
+
* Default 200; 0 forwards every delta as it arrives.
|
|
135
|
+
*/
|
|
136
|
+
playoutLeadMs?: number;
|
|
129
137
|
/** Speech gate tuning (utterance boundaries synthesized from the audio). */
|
|
130
138
|
speechGate?: SpeechGateOptions;
|
|
131
139
|
/** Session-timeline gap that splits transcript fragments into turns. Default 800. */
|
|
@@ -137,6 +145,7 @@ interface GptLiveProviderConfig {
|
|
|
137
145
|
ackTimeoutMs?: number;
|
|
138
146
|
providerName?: string;
|
|
139
147
|
}
|
|
148
|
+
declare const DEFAULT_PLAYOUT_LEAD_MS = 200;
|
|
140
149
|
declare class GptLiveProvider extends BaseRealtimeProvider {
|
|
141
150
|
readonly name: string;
|
|
142
151
|
readonly capabilities: ProviderCapabilities;
|
|
@@ -152,6 +161,10 @@ declare class GptLiveProvider extends BaseRealtimeProvider {
|
|
|
152
161
|
expiresAt: number | null;
|
|
153
162
|
private readonly gate;
|
|
154
163
|
private gateStallTimer;
|
|
164
|
+
/** Deltas of the current utterance held until the playout lead has accumulated. */
|
|
165
|
+
private lead;
|
|
166
|
+
/** The last idle delta: becomes the utterance's pre-roll when the gate opens. */
|
|
167
|
+
private preRoll;
|
|
155
168
|
private utteranceCounter;
|
|
156
169
|
private currentUtteranceId;
|
|
157
170
|
private lastUtteranceId;
|
|
@@ -192,6 +205,18 @@ declare class GptLiveProvider extends BaseRealtimeProvider {
|
|
|
192
205
|
updateSession(patch: Partial<ProviderSessionInit>, options?: SessionUpdateOptions): Promise<boolean>;
|
|
193
206
|
private handleEvent;
|
|
194
207
|
private handleOutputAudio;
|
|
208
|
+
/**
|
|
209
|
+
* The stream is real-time paced, so Twilio's buffer never runs ahead of
|
|
210
|
+
* playout and every delivery hiccup between the model, this server and
|
|
211
|
+
* Twilio is an audible gap (field, Sept 2026: "slightly choppy"). Hold the
|
|
212
|
+
* first `playoutLeadMs` of each utterance — the last idle delta included —
|
|
213
|
+
* then flush and stream through: Twilio keeps that much cushion for the
|
|
214
|
+
* rest of the utterance, at the cost of that much latency on its first word.
|
|
215
|
+
*/
|
|
216
|
+
private armPlayoutLead;
|
|
217
|
+
private forwardAudio;
|
|
218
|
+
private flushPlayoutLead;
|
|
219
|
+
private playoutLeadMs;
|
|
195
220
|
private handleBackendEvent;
|
|
196
221
|
private beginUtterance;
|
|
197
222
|
private endUtterance;
|
|
@@ -244,6 +269,14 @@ interface GptLiveOptions {
|
|
|
244
269
|
/** Provider-native `session.start` fields, deep-merged last. The schema is strict: an unknown field rejects the session. */
|
|
245
270
|
sessionOptions?: Record<string, unknown>;
|
|
246
271
|
connectTimeoutMs?: number;
|
|
272
|
+
/**
|
|
273
|
+
* Cushion held at the start of each utterance before audio is forwarded to
|
|
274
|
+
* Twilio. The model streams at exactly real-time pace, so without it any
|
|
275
|
+
* delivery hiccup is an audible gap; with it Twilio stays that far ahead of
|
|
276
|
+
* playout. Costs the same amount of latency on each turn's first word.
|
|
277
|
+
* Default 200; 0 forwards every delta as it arrives.
|
|
278
|
+
*/
|
|
279
|
+
playoutLeadMs?: number;
|
|
247
280
|
/** Speech gate tuning — utterance boundaries synthesized from the continuous output stream. */
|
|
248
281
|
speechGate?: SpeechGateOptions;
|
|
249
282
|
/** Session-timeline gap that splits transcript fragments into turns. Default 800. */
|
|
@@ -258,4 +291,4 @@ interface GptLiveOptions {
|
|
|
258
291
|
*/
|
|
259
292
|
declare function gptLive(options?: GptLiveOptions): ProviderFactory;
|
|
260
293
|
//#endregion
|
|
261
|
-
export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, type GptLiveDelegationOptions, GptLiveOptions, GptLiveProvider, type GptLiveProviderConfig, type GptLiveSessionConfig, GptLiveVoice, SpeechGate, type SpeechGateOptions, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
|
|
294
|
+
export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, DEFAULT_PLAYOUT_LEAD_MS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, type GptLiveDelegationOptions, GptLiveOptions, GptLiveProvider, type GptLiveProviderConfig, type GptLiveSessionConfig, GptLiveVoice, SpeechGate, type SpeechGateOptions, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
|
package/dist/gpt-live.d.mts
CHANGED
|
@@ -126,6 +126,14 @@ interface GptLiveProviderConfig {
|
|
|
126
126
|
connectTimeoutMs?: number;
|
|
127
127
|
/** How long `close()` waits for `session.closed` (final usage) before dropping the socket. Default 3000. */
|
|
128
128
|
closeTimeoutMs?: number;
|
|
129
|
+
/**
|
|
130
|
+
* Audio held back at the start of each utterance before forwarding begins,
|
|
131
|
+
* so Twilio keeps that much cushion against delivery jitter (the stream is
|
|
132
|
+
* real-time paced — without it any hiccup is an audible gap). The last idle
|
|
133
|
+
* delta before the onset is included, so soft onsets are not clipped.
|
|
134
|
+
* Default 200; 0 forwards every delta as it arrives.
|
|
135
|
+
*/
|
|
136
|
+
playoutLeadMs?: number;
|
|
129
137
|
/** Speech gate tuning (utterance boundaries synthesized from the audio). */
|
|
130
138
|
speechGate?: SpeechGateOptions;
|
|
131
139
|
/** Session-timeline gap that splits transcript fragments into turns. Default 800. */
|
|
@@ -137,6 +145,7 @@ interface GptLiveProviderConfig {
|
|
|
137
145
|
ackTimeoutMs?: number;
|
|
138
146
|
providerName?: string;
|
|
139
147
|
}
|
|
148
|
+
declare const DEFAULT_PLAYOUT_LEAD_MS = 200;
|
|
140
149
|
declare class GptLiveProvider extends BaseRealtimeProvider {
|
|
141
150
|
readonly name: string;
|
|
142
151
|
readonly capabilities: ProviderCapabilities;
|
|
@@ -152,6 +161,10 @@ declare class GptLiveProvider extends BaseRealtimeProvider {
|
|
|
152
161
|
expiresAt: number | null;
|
|
153
162
|
private readonly gate;
|
|
154
163
|
private gateStallTimer;
|
|
164
|
+
/** Deltas of the current utterance held until the playout lead has accumulated. */
|
|
165
|
+
private lead;
|
|
166
|
+
/** The last idle delta: becomes the utterance's pre-roll when the gate opens. */
|
|
167
|
+
private preRoll;
|
|
155
168
|
private utteranceCounter;
|
|
156
169
|
private currentUtteranceId;
|
|
157
170
|
private lastUtteranceId;
|
|
@@ -192,6 +205,18 @@ declare class GptLiveProvider extends BaseRealtimeProvider {
|
|
|
192
205
|
updateSession(patch: Partial<ProviderSessionInit>, options?: SessionUpdateOptions): Promise<boolean>;
|
|
193
206
|
private handleEvent;
|
|
194
207
|
private handleOutputAudio;
|
|
208
|
+
/**
|
|
209
|
+
* The stream is real-time paced, so Twilio's buffer never runs ahead of
|
|
210
|
+
* playout and every delivery hiccup between the model, this server and
|
|
211
|
+
* Twilio is an audible gap (field, Sept 2026: "slightly choppy"). Hold the
|
|
212
|
+
* first `playoutLeadMs` of each utterance — the last idle delta included —
|
|
213
|
+
* then flush and stream through: Twilio keeps that much cushion for the
|
|
214
|
+
* rest of the utterance, at the cost of that much latency on its first word.
|
|
215
|
+
*/
|
|
216
|
+
private armPlayoutLead;
|
|
217
|
+
private forwardAudio;
|
|
218
|
+
private flushPlayoutLead;
|
|
219
|
+
private playoutLeadMs;
|
|
195
220
|
private handleBackendEvent;
|
|
196
221
|
private beginUtterance;
|
|
197
222
|
private endUtterance;
|
|
@@ -244,6 +269,14 @@ interface GptLiveOptions {
|
|
|
244
269
|
/** Provider-native `session.start` fields, deep-merged last. The schema is strict: an unknown field rejects the session. */
|
|
245
270
|
sessionOptions?: Record<string, unknown>;
|
|
246
271
|
connectTimeoutMs?: number;
|
|
272
|
+
/**
|
|
273
|
+
* Cushion held at the start of each utterance before audio is forwarded to
|
|
274
|
+
* Twilio. The model streams at exactly real-time pace, so without it any
|
|
275
|
+
* delivery hiccup is an audible gap; with it Twilio stays that far ahead of
|
|
276
|
+
* playout. Costs the same amount of latency on each turn's first word.
|
|
277
|
+
* Default 200; 0 forwards every delta as it arrives.
|
|
278
|
+
*/
|
|
279
|
+
playoutLeadMs?: number;
|
|
247
280
|
/** Speech gate tuning — utterance boundaries synthesized from the continuous output stream. */
|
|
248
281
|
speechGate?: SpeechGateOptions;
|
|
249
282
|
/** Session-timeline gap that splits transcript fragments into turns. Default 800. */
|
|
@@ -258,4 +291,4 @@ interface GptLiveOptions {
|
|
|
258
291
|
*/
|
|
259
292
|
declare function gptLive(options?: GptLiveOptions): ProviderFactory;
|
|
260
293
|
//#endregion
|
|
261
|
-
export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, type GptLiveDelegationOptions, GptLiveOptions, GptLiveProvider, type GptLiveProviderConfig, type GptLiveSessionConfig, GptLiveVoice, SpeechGate, type SpeechGateOptions, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
|
|
294
|
+
export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, DEFAULT_PLAYOUT_LEAD_MS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, type GptLiveDelegationOptions, GptLiveOptions, GptLiveProvider, type GptLiveProviderConfig, type GptLiveSessionConfig, GptLiveVoice, SpeechGate, type SpeechGateOptions, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
|
package/dist/gpt-live.mjs
CHANGED
|
@@ -307,6 +307,8 @@ const NON_RETRIABLE_CLOSE_CODES = /* @__PURE__ */ new Set([
|
|
|
307
307
|
const APPEND_MAX_CHARS = 1200;
|
|
308
308
|
const TRANSCRIPT_IDLE_EXTRA_MS = 300;
|
|
309
309
|
const GATE_STALL_EXTRA_MS = 500;
|
|
310
|
+
const DEFAULT_PLAYOUT_LEAD_MS = 200;
|
|
311
|
+
const MULAW_BYTES_PER_MS = 8;
|
|
310
312
|
const ACK_TYPES = {
|
|
311
313
|
"session.instructions.append": "session.instructions.appended",
|
|
312
314
|
"session.thinking.append": "session.thinking.appended",
|
|
@@ -343,6 +345,10 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
|
|
|
343
345
|
expiresAt = null;
|
|
344
346
|
gate;
|
|
345
347
|
gateStallTimer = null;
|
|
348
|
+
/** Deltas of the current utterance held until the playout lead has accumulated. */
|
|
349
|
+
lead = null;
|
|
350
|
+
/** The last idle delta: becomes the utterance's pre-roll when the gate opens. */
|
|
351
|
+
preRoll = null;
|
|
346
352
|
utteranceCounter = 0;
|
|
347
353
|
currentUtteranceId = null;
|
|
348
354
|
lastUtteranceId = null;
|
|
@@ -489,6 +495,8 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
|
|
|
489
495
|
this.inputTranscript.dispose();
|
|
490
496
|
this.outputTranscript.dispose();
|
|
491
497
|
this.clearGateStall();
|
|
498
|
+
this.lead = null;
|
|
499
|
+
this.preRoll = null;
|
|
492
500
|
if (!ws || ws.readyState === WebSocket$1.CLOSED) return;
|
|
493
501
|
await new Promise((resolve) => {
|
|
494
502
|
const timer = setTimeout(() => finish(), this.config.closeTimeoutMs ?? 3e3);
|
|
@@ -682,14 +690,63 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
|
|
|
682
690
|
for (const gateEvent of events) if (gateEvent.type === "open") {
|
|
683
691
|
this.beginUtterance();
|
|
684
692
|
forwardId = this.currentUtteranceId;
|
|
693
|
+
this.armPlayoutLead();
|
|
685
694
|
} else close = gateEvent;
|
|
686
|
-
if (forwardId) this.
|
|
687
|
-
|
|
688
|
-
|
|
689
|
-
|
|
695
|
+
if (forwardId) this.forwardAudio(delta, bytes.length / MULAW_BYTES_PER_MS, forwardId);
|
|
696
|
+
else this.preRoll = {
|
|
697
|
+
delta,
|
|
698
|
+
ms: bytes.length / MULAW_BYTES_PER_MS
|
|
699
|
+
};
|
|
690
700
|
if (close) this.endUtterance();
|
|
691
701
|
else if (this.gate.isOpen) this.armGateStall();
|
|
692
702
|
}
|
|
703
|
+
/**
|
|
704
|
+
* The stream is real-time paced, so Twilio's buffer never runs ahead of
|
|
705
|
+
* playout and every delivery hiccup between the model, this server and
|
|
706
|
+
* Twilio is an audible gap (field, Sept 2026: "slightly choppy"). Hold the
|
|
707
|
+
* first `playoutLeadMs` of each utterance — the last idle delta included —
|
|
708
|
+
* then flush and stream through: Twilio keeps that much cushion for the
|
|
709
|
+
* rest of the utterance, at the cost of that much latency on its first word.
|
|
710
|
+
*/
|
|
711
|
+
armPlayoutLead() {
|
|
712
|
+
if (this.playoutLeadMs() <= 0) {
|
|
713
|
+
this.preRoll = null;
|
|
714
|
+
return;
|
|
715
|
+
}
|
|
716
|
+
this.lead = {
|
|
717
|
+
pending: [],
|
|
718
|
+
pendingMs: 0
|
|
719
|
+
};
|
|
720
|
+
if (this.preRoll) {
|
|
721
|
+
this.lead.pending.push(this.preRoll.delta);
|
|
722
|
+
this.lead.pendingMs += this.preRoll.ms;
|
|
723
|
+
this.preRoll = null;
|
|
724
|
+
}
|
|
725
|
+
}
|
|
726
|
+
forwardAudio(delta, ms, responseId) {
|
|
727
|
+
if (!this.lead) {
|
|
728
|
+
this.emit("audio", {
|
|
729
|
+
base64Mulaw: delta,
|
|
730
|
+
responseId
|
|
731
|
+
});
|
|
732
|
+
return;
|
|
733
|
+
}
|
|
734
|
+
this.lead.pending.push(delta);
|
|
735
|
+
this.lead.pendingMs += ms;
|
|
736
|
+
if (this.lead.pendingMs >= this.playoutLeadMs()) this.flushPlayoutLead(responseId);
|
|
737
|
+
}
|
|
738
|
+
flushPlayoutLead(responseId) {
|
|
739
|
+
const lead = this.lead;
|
|
740
|
+
if (!lead) return;
|
|
741
|
+
this.lead = null;
|
|
742
|
+
for (const delta of lead.pending) this.emit("audio", {
|
|
743
|
+
base64Mulaw: delta,
|
|
744
|
+
responseId
|
|
745
|
+
});
|
|
746
|
+
}
|
|
747
|
+
playoutLeadMs() {
|
|
748
|
+
return this.config.playoutLeadMs ?? 200;
|
|
749
|
+
}
|
|
693
750
|
handleBackendEvent(inner, delegationId) {
|
|
694
751
|
const type = inner.type ?? "";
|
|
695
752
|
if (type === "response.output_item.done" && inner.item?.type === "function_call") {
|
|
@@ -726,6 +783,7 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
|
|
|
726
783
|
endUtterance() {
|
|
727
784
|
this.clearGateStall();
|
|
728
785
|
const id = this.currentUtteranceId;
|
|
786
|
+
if (id) this.flushPlayoutLead(id);
|
|
729
787
|
this.currentUtteranceId = null;
|
|
730
788
|
if (id) this.emit("responseDone", { responseId: id });
|
|
731
789
|
}
|
|
@@ -754,6 +812,8 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
|
|
|
754
812
|
resetUtteranceState() {
|
|
755
813
|
this.clearGateStall();
|
|
756
814
|
this.gate.close();
|
|
815
|
+
this.lead = null;
|
|
816
|
+
this.preRoll = null;
|
|
757
817
|
this.currentUtteranceId = null;
|
|
758
818
|
this.lastUtteranceId = null;
|
|
759
819
|
this.seenCalls.clear();
|
|
@@ -920,6 +980,7 @@ function gptLive(options = {}) {
|
|
|
920
980
|
store: options.store,
|
|
921
981
|
extraSessionOptions: options.sessionOptions,
|
|
922
982
|
connectTimeoutMs: options.connectTimeoutMs,
|
|
983
|
+
playoutLeadMs: options.playoutLeadMs,
|
|
923
984
|
speechGate: options.speechGate,
|
|
924
985
|
transcriptGapMs: options.transcriptGapMs
|
|
925
986
|
};
|
|
@@ -933,4 +994,4 @@ function gptLive(options = {}) {
|
|
|
933
994
|
};
|
|
934
995
|
}
|
|
935
996
|
//#endregion
|
|
936
|
-
export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, GptLiveProvider, SpeechGate, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
|
|
997
|
+
export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, DEFAULT_PLAYOUT_LEAD_MS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, GptLiveProvider, SpeechGate, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
|
package/dist/index.cjs
CHANGED
|
@@ -535,6 +535,28 @@ var PlaybackTracker = class {
|
|
|
535
535
|
return interrupted;
|
|
536
536
|
}
|
|
537
537
|
/**
|
|
538
|
+
* The provider session behind every open response is gone (closed for a
|
|
539
|
+
* handoff or reconnect), so their remaining marks may never come home:
|
|
540
|
+
* finalize them at their current playedMs so `isPlaybackActive()` cannot
|
|
541
|
+
* stay true for the rest of the call. Late echoes classify as flushed.
|
|
542
|
+
* Returns the abandoned responses.
|
|
543
|
+
*/
|
|
544
|
+
abandonOpen() {
|
|
545
|
+
const abandoned = [];
|
|
546
|
+
for (const track of [...this.responses.values()]) {
|
|
547
|
+
if (track.finished) continue;
|
|
548
|
+
track.flushed = true;
|
|
549
|
+
track.finished = true;
|
|
550
|
+
abandoned.push({
|
|
551
|
+
responseId: track.responseId,
|
|
552
|
+
playedMs: track.playedMs,
|
|
553
|
+
itemId: track.itemId
|
|
554
|
+
});
|
|
555
|
+
this.maybeForget(track);
|
|
556
|
+
}
|
|
557
|
+
return abandoned;
|
|
558
|
+
}
|
|
559
|
+
/**
|
|
538
560
|
* Best-estimate of what the caller has heard of `responseId` right now:
|
|
539
561
|
* last confirmed mark plus wall-clock elapsed since, clamped to the total.
|
|
540
562
|
*/
|
|
@@ -1099,6 +1121,8 @@ var CallSession = class extends require_events.TypedEmitter {
|
|
|
1099
1121
|
reconnecting = false;
|
|
1100
1122
|
pendingHangup = null;
|
|
1101
1123
|
pendingTransfer = null;
|
|
1124
|
+
/** Model-owned turn-taking: a transfer that landed mid-utterance, held until the sentence plays out. */
|
|
1125
|
+
pendingHandoff = null;
|
|
1102
1126
|
endedReason = null;
|
|
1103
1127
|
hangupReason = "agent-hangup";
|
|
1104
1128
|
middlewares;
|
|
@@ -1491,6 +1515,11 @@ var CallSession = class extends require_events.TypedEmitter {
|
|
|
1491
1515
|
this.pendingHangup.grace = null;
|
|
1492
1516
|
}
|
|
1493
1517
|
}
|
|
1518
|
+
if (this.pendingHandoff?.grace) {
|
|
1519
|
+
clearTimeout(this.pendingHandoff.grace);
|
|
1520
|
+
this.timers.delete(this.pendingHandoff.grace);
|
|
1521
|
+
this.pendingHandoff.grace = null;
|
|
1522
|
+
}
|
|
1494
1523
|
this.noteHangupProgress();
|
|
1495
1524
|
this.clearIdleTimer();
|
|
1496
1525
|
this.interruptions.onResponseStarted(responseId);
|
|
@@ -1511,6 +1540,7 @@ var CallSession = class extends require_events.TypedEmitter {
|
|
|
1511
1540
|
this.flushToolQueue();
|
|
1512
1541
|
this.executePendingTransfer();
|
|
1513
1542
|
this.maybeCompleteHangup();
|
|
1543
|
+
this.maybeRunPendingHandoff();
|
|
1514
1544
|
}
|
|
1515
1545
|
});
|
|
1516
1546
|
provider.on("agentTranscript", ({ responseId, text }) => {
|
|
@@ -1578,7 +1608,7 @@ var CallSession = class extends require_events.TypedEmitter {
|
|
|
1578
1608
|
const proposal = this.noiseVad.onInboundFrame(payload, excluded);
|
|
1579
1609
|
if (proposal) this.handleVadProposal(proposal);
|
|
1580
1610
|
}
|
|
1581
|
-
if (this.pregreeting !== null && !this.pregreeting.played || this.
|
|
1611
|
+
if (this.pregreeting !== null && !this.pregreeting.played || this.firstTurnDeafness() && !this.firstTurnDone || this.deps.options.deafness.muteDuringToolExecution && this.runningTools.size > 0 || this.deps.options.deafness.muteWhileAgentSpeaking && this.tracker.isPlaybackActive() || this.interruptions.isSuspended) {
|
|
1582
1612
|
if (!this.modelOwnsTurnTaking()) return;
|
|
1583
1613
|
payload = silenceLike(payload);
|
|
1584
1614
|
}
|
|
@@ -1673,6 +1703,7 @@ var CallSession = class extends require_events.TypedEmitter {
|
|
|
1673
1703
|
this.flushToolQueue();
|
|
1674
1704
|
this.executePendingTransfer();
|
|
1675
1705
|
this.maybeCompleteHangup();
|
|
1706
|
+
this.maybeRunPendingHandoff();
|
|
1676
1707
|
if (this.blockedUserTurn === "committed") {
|
|
1677
1708
|
this.blockedUserTurn = "idle";
|
|
1678
1709
|
if (!hadQueuedToolResults && this.runningTools.size === 0) this.provider?.createResponse();
|
|
@@ -1862,6 +1893,10 @@ var CallSession = class extends require_events.TypedEmitter {
|
|
|
1862
1893
|
result: { handoffTo: target.id },
|
|
1863
1894
|
durationMs: Date.now() - started
|
|
1864
1895
|
});
|
|
1896
|
+
if (this.modelOwnsTurnTaking() && (this.generating || this.tracker.isPlaybackActive())) {
|
|
1897
|
+
this.deferHandoff(target, directive.reason);
|
|
1898
|
+
return;
|
|
1899
|
+
}
|
|
1865
1900
|
await this.performHandoff(target, directive.reason);
|
|
1866
1901
|
return;
|
|
1867
1902
|
}
|
|
@@ -2162,7 +2197,7 @@ var CallSession = class extends require_events.TypedEmitter {
|
|
|
2162
2197
|
pending.grace = null;
|
|
2163
2198
|
if (this.generating || this.tracker.isPlaybackActive()) return;
|
|
2164
2199
|
this.completeHangup();
|
|
2165
|
-
},
|
|
2200
|
+
}, SENTENCE_GRACE_MS);
|
|
2166
2201
|
timer.unref?.();
|
|
2167
2202
|
this.timers.add(timer);
|
|
2168
2203
|
this.pendingHangup.grace = timer;
|
|
@@ -2229,6 +2264,7 @@ var CallSession = class extends require_events.TypedEmitter {
|
|
|
2229
2264
|
if (this.stateValue !== "active") return;
|
|
2230
2265
|
const provider = this.provider;
|
|
2231
2266
|
if (!provider) return;
|
|
2267
|
+
this.abandonOpenPlayback();
|
|
2232
2268
|
try {
|
|
2233
2269
|
await provider.connect(this.buildProviderInit());
|
|
2234
2270
|
} catch (error) {
|
|
@@ -2277,6 +2313,89 @@ var CallSession = class extends require_events.TypedEmitter {
|
|
|
2277
2313
|
cause: "no-caller-turn"
|
|
2278
2314
|
});
|
|
2279
2315
|
}
|
|
2316
|
+
/**
|
|
2317
|
+
* First-turn deafness shields an agent-first greeting from early caller
|
|
2318
|
+
* speech where the bridge owns barge-in. A full-duplex model owns talk-over
|
|
2319
|
+
* itself, and "deaf" there only means fed silence — so the default flips
|
|
2320
|
+
* off on `turnTaking: 'model'`; an explicit setting is honored as written.
|
|
2321
|
+
*/
|
|
2322
|
+
firstTurnDeafness() {
|
|
2323
|
+
return this.deps.options.deafness.ignoreUserAudioUntilFirstTurnDone ?? !this.modelOwnsTurnTaking();
|
|
2324
|
+
}
|
|
2325
|
+
/**
|
|
2326
|
+
* The provider session behind every open response is being closed (handoff
|
|
2327
|
+
* or reconnect): their tail marks may never come. Finalize them now, or
|
|
2328
|
+
* `isPlaybackActive()` stays true for the rest of the call and everything
|
|
2329
|
+
* gated on it — the hangup grace, REST transfers, guard rotation — wedges
|
|
2330
|
+
* (field, Sept 2026). Whatever Twilio still holds plays out; no
|
|
2331
|
+
* `playback.finished` is claimed for audio nobody confirmed.
|
|
2332
|
+
*/
|
|
2333
|
+
abandonOpenPlayback() {
|
|
2334
|
+
if (this.tracker.abandonOpen().length === 0) return;
|
|
2335
|
+
this.interruptions.onPlaybackEnded();
|
|
2336
|
+
this.maybeCompleteHangup();
|
|
2337
|
+
this.maybeRunPendingHandoff();
|
|
2338
|
+
}
|
|
2339
|
+
/**
|
|
2340
|
+
* On a full-duplex provider the backend's transfer lands while the voice is
|
|
2341
|
+
* still speaking the sentence that announces it, and the handoff is a
|
|
2342
|
+
* close-and-reopen: performing it at once cuts that sentence, and the
|
|
2343
|
+
* in-flight deltas kill the handoff hold before it starts (field, Sept
|
|
2344
|
+
* 2026). Hold the transfer until the utterance has played out plus one
|
|
2345
|
+
* sentence gap — a new utterance cancels the grace — capped so a voice that
|
|
2346
|
+
* never stops still hands off.
|
|
2347
|
+
*/
|
|
2348
|
+
deferHandoff(target, reason) {
|
|
2349
|
+
if (this.pendingHandoff) {
|
|
2350
|
+
this.pendingHandoff.target = target;
|
|
2351
|
+
this.pendingHandoff.reason = reason;
|
|
2352
|
+
return;
|
|
2353
|
+
}
|
|
2354
|
+
const cap = setTimeout(() => {
|
|
2355
|
+
this.timers.delete(cap);
|
|
2356
|
+
const pending = this.pendingHandoff;
|
|
2357
|
+
if (!pending) return;
|
|
2358
|
+
this.log.warn("deferred handoff capped — the voice kept talking; handing off now");
|
|
2359
|
+
this.clearPendingHandoff();
|
|
2360
|
+
this.performHandoff(pending.target, pending.reason);
|
|
2361
|
+
}, HANDOFF_DEFER_MAX_MS);
|
|
2362
|
+
cap.unref?.();
|
|
2363
|
+
this.timers.add(cap);
|
|
2364
|
+
this.pendingHandoff = {
|
|
2365
|
+
target,
|
|
2366
|
+
reason,
|
|
2367
|
+
grace: null,
|
|
2368
|
+
cap
|
|
2369
|
+
};
|
|
2370
|
+
this.maybeRunPendingHandoff();
|
|
2371
|
+
}
|
|
2372
|
+
maybeRunPendingHandoff() {
|
|
2373
|
+
const pending = this.pendingHandoff;
|
|
2374
|
+
if (!pending || pending.grace) return;
|
|
2375
|
+
if (this.generating || this.tracker.isPlaybackActive()) return;
|
|
2376
|
+
const grace = setTimeout(() => {
|
|
2377
|
+
this.timers.delete(grace);
|
|
2378
|
+
const current = this.pendingHandoff;
|
|
2379
|
+
if (!current) return;
|
|
2380
|
+
current.grace = null;
|
|
2381
|
+
if (this.generating || this.tracker.isPlaybackActive()) return;
|
|
2382
|
+
this.clearPendingHandoff();
|
|
2383
|
+
this.performHandoff(current.target, current.reason);
|
|
2384
|
+
}, SENTENCE_GRACE_MS);
|
|
2385
|
+
grace.unref?.();
|
|
2386
|
+
this.timers.add(grace);
|
|
2387
|
+
pending.grace = grace;
|
|
2388
|
+
}
|
|
2389
|
+
clearPendingHandoff() {
|
|
2390
|
+
const pending = this.pendingHandoff;
|
|
2391
|
+
if (!pending) return;
|
|
2392
|
+
this.pendingHandoff = null;
|
|
2393
|
+
for (const timer of [pending.grace, pending.cap]) {
|
|
2394
|
+
if (!timer) continue;
|
|
2395
|
+
clearTimeout(timer);
|
|
2396
|
+
this.timers.delete(timer);
|
|
2397
|
+
}
|
|
2398
|
+
}
|
|
2280
2399
|
async performHandoff(target, reason) {
|
|
2281
2400
|
if (this.stateValue !== "active" || !this.provider) return;
|
|
2282
2401
|
if (target.id === this.activeAgentValue.id) return;
|
|
@@ -2309,6 +2428,7 @@ var CallSession = class extends require_events.TypedEmitter {
|
|
|
2309
2428
|
startDelayMs: hold.startDelayMs ?? 300
|
|
2310
2429
|
});
|
|
2311
2430
|
try {
|
|
2431
|
+
this.abandonOpenPlayback();
|
|
2312
2432
|
await this.provider.close();
|
|
2313
2433
|
await this.provider.connect({
|
|
2314
2434
|
...this.buildProviderInit(),
|
|
@@ -2509,7 +2629,10 @@ var CallSession = class extends require_events.TypedEmitter {
|
|
|
2509
2629
|
};
|
|
2510
2630
|
const PREGREETING_MARK = "pre:greeting";
|
|
2511
2631
|
/** Longer than the speech gate's quiet window plus the mark round trip (0.8 s + ~0.3 s). */
|
|
2512
|
-
|
|
2632
|
+
/** Model-owned turn-taking: the gate can split one sentence at a pause — wait one gap for the next. */
|
|
2633
|
+
const SENTENCE_GRACE_MS = 1500;
|
|
2634
|
+
/** A deferred handoff runs no later than this after the transfer landed, even mid-utterance. */
|
|
2635
|
+
const HANDOFF_DEFER_MAX_MS = 5e3;
|
|
2513
2636
|
/** 400ms per frame — matches production burst-write implementations. */
|
|
2514
2637
|
const PREGREETING_CHUNK_BYTES = 3200;
|
|
2515
2638
|
function toError(value) {
|
|
@@ -2536,7 +2659,6 @@ const DEFAULT_SESSION_OPTIONS = {
|
|
|
2536
2659
|
greeting: { mode: "agent-initiates" },
|
|
2537
2660
|
interruptions: { enabled: true },
|
|
2538
2661
|
deafness: {
|
|
2539
|
-
ignoreUserAudioUntilFirstTurnDone: true,
|
|
2540
2662
|
muteDuringToolExecution: true,
|
|
2541
2663
|
muteWhileAgentSpeaking: false
|
|
2542
2664
|
},
|
package/dist/index.d.cts
CHANGED
|
@@ -521,11 +521,14 @@ interface GreetingOptions {
|
|
|
521
521
|
interface DeafnessOptions {
|
|
522
522
|
/**
|
|
523
523
|
* Drop caller audio until the agent's first turn finishes playing.
|
|
524
|
-
* Protects the greeting from noisy pickups
|
|
525
|
-
*
|
|
526
|
-
*
|
|
527
|
-
*
|
|
528
|
-
*
|
|
524
|
+
* Protects the greeting from noisy pickups on providers where the bridge
|
|
525
|
+
* owns barge-in. Unset = automatic: true there, false with
|
|
526
|
+
* `greeting.mode: 'user-initiates'` (the caller must be heard to start the
|
|
527
|
+
* call at all) and on full-duplex providers (`turnTaking: 'model'`), where
|
|
528
|
+
* the model owns talk-over and "deaf" only means fed silence. An explicit
|
|
529
|
+
* value is honored as written — an explicit true with user-initiates
|
|
530
|
+
* deafens the call until something else (an idle nudge, a tool) produces
|
|
531
|
+
* the agent's first turn.
|
|
529
532
|
*/
|
|
530
533
|
ignoreUserAudioUntilFirstTurnDone?: boolean;
|
|
531
534
|
/** Drop caller audio while a foreground tool is running. Default true. */
|
|
@@ -914,6 +917,8 @@ declare class CallSession extends TypedEmitter<SessionEventMap> {
|
|
|
914
917
|
private reconnecting;
|
|
915
918
|
private pendingHangup;
|
|
916
919
|
private pendingTransfer;
|
|
920
|
+
/** Model-owned turn-taking: a transfer that landed mid-utterance, held until the sentence plays out. */
|
|
921
|
+
private pendingHandoff;
|
|
917
922
|
private endedReason;
|
|
918
923
|
private hangupReason;
|
|
919
924
|
private readonly middlewares;
|
|
@@ -1098,6 +1103,34 @@ declare class CallSession extends TypedEmitter<SessionEventMap> {
|
|
|
1098
1103
|
* calls the same transfer tool again on its next turn.
|
|
1099
1104
|
*/
|
|
1100
1105
|
private rejectHandoff;
|
|
1106
|
+
/**
|
|
1107
|
+
* First-turn deafness shields an agent-first greeting from early caller
|
|
1108
|
+
* speech where the bridge owns barge-in. A full-duplex model owns talk-over
|
|
1109
|
+
* itself, and "deaf" there only means fed silence — so the default flips
|
|
1110
|
+
* off on `turnTaking: 'model'`; an explicit setting is honored as written.
|
|
1111
|
+
*/
|
|
1112
|
+
private firstTurnDeafness;
|
|
1113
|
+
/**
|
|
1114
|
+
* The provider session behind every open response is being closed (handoff
|
|
1115
|
+
* or reconnect): their tail marks may never come. Finalize them now, or
|
|
1116
|
+
* `isPlaybackActive()` stays true for the rest of the call and everything
|
|
1117
|
+
* gated on it — the hangup grace, REST transfers, guard rotation — wedges
|
|
1118
|
+
* (field, Sept 2026). Whatever Twilio still holds plays out; no
|
|
1119
|
+
* `playback.finished` is claimed for audio nobody confirmed.
|
|
1120
|
+
*/
|
|
1121
|
+
private abandonOpenPlayback;
|
|
1122
|
+
/**
|
|
1123
|
+
* On a full-duplex provider the backend's transfer lands while the voice is
|
|
1124
|
+
* still speaking the sentence that announces it, and the handoff is a
|
|
1125
|
+
* close-and-reopen: performing it at once cuts that sentence, and the
|
|
1126
|
+
* in-flight deltas kill the handoff hold before it starts (field, Sept
|
|
1127
|
+
* 2026). Hold the transfer until the utterance has played out plus one
|
|
1128
|
+
* sentence gap — a new utterance cancels the grace — capped so a voice that
|
|
1129
|
+
* never stops still hands off.
|
|
1130
|
+
*/
|
|
1131
|
+
private deferHandoff;
|
|
1132
|
+
private maybeRunPendingHandoff;
|
|
1133
|
+
private clearPendingHandoff;
|
|
1101
1134
|
private performHandoff;
|
|
1102
1135
|
/**
|
|
1103
1136
|
* Burst-write a stored μ-law greeting straight onto the Twilio socket —
|
|
@@ -1276,6 +1309,18 @@ declare class PlaybackTracker {
|
|
|
1276
1309
|
playedMs: number;
|
|
1277
1310
|
itemId?: string;
|
|
1278
1311
|
}>;
|
|
1312
|
+
/**
|
|
1313
|
+
* The provider session behind every open response is gone (closed for a
|
|
1314
|
+
* handoff or reconnect), so their remaining marks may never come home:
|
|
1315
|
+
* finalize them at their current playedMs so `isPlaybackActive()` cannot
|
|
1316
|
+
* stay true for the rest of the call. Late echoes classify as flushed.
|
|
1317
|
+
* Returns the abandoned responses.
|
|
1318
|
+
*/
|
|
1319
|
+
abandonOpen(): Array<{
|
|
1320
|
+
responseId: string;
|
|
1321
|
+
playedMs: number;
|
|
1322
|
+
itemId?: string;
|
|
1323
|
+
}>;
|
|
1279
1324
|
/**
|
|
1280
1325
|
* Best-estimate of what the caller has heard of `responseId` right now:
|
|
1281
1326
|
* last confirmed mark plus wall-clock elapsed since, clamped to the total.
|
package/dist/index.d.mts
CHANGED
|
@@ -521,11 +521,14 @@ interface GreetingOptions {
|
|
|
521
521
|
interface DeafnessOptions {
|
|
522
522
|
/**
|
|
523
523
|
* Drop caller audio until the agent's first turn finishes playing.
|
|
524
|
-
* Protects the greeting from noisy pickups
|
|
525
|
-
*
|
|
526
|
-
*
|
|
527
|
-
*
|
|
528
|
-
*
|
|
524
|
+
* Protects the greeting from noisy pickups on providers where the bridge
|
|
525
|
+
* owns barge-in. Unset = automatic: true there, false with
|
|
526
|
+
* `greeting.mode: 'user-initiates'` (the caller must be heard to start the
|
|
527
|
+
* call at all) and on full-duplex providers (`turnTaking: 'model'`), where
|
|
528
|
+
* the model owns talk-over and "deaf" only means fed silence. An explicit
|
|
529
|
+
* value is honored as written — an explicit true with user-initiates
|
|
530
|
+
* deafens the call until something else (an idle nudge, a tool) produces
|
|
531
|
+
* the agent's first turn.
|
|
529
532
|
*/
|
|
530
533
|
ignoreUserAudioUntilFirstTurnDone?: boolean;
|
|
531
534
|
/** Drop caller audio while a foreground tool is running. Default true. */
|
|
@@ -914,6 +917,8 @@ declare class CallSession extends TypedEmitter<SessionEventMap> {
|
|
|
914
917
|
private reconnecting;
|
|
915
918
|
private pendingHangup;
|
|
916
919
|
private pendingTransfer;
|
|
920
|
+
/** Model-owned turn-taking: a transfer that landed mid-utterance, held until the sentence plays out. */
|
|
921
|
+
private pendingHandoff;
|
|
917
922
|
private endedReason;
|
|
918
923
|
private hangupReason;
|
|
919
924
|
private readonly middlewares;
|
|
@@ -1098,6 +1103,34 @@ declare class CallSession extends TypedEmitter<SessionEventMap> {
|
|
|
1098
1103
|
* calls the same transfer tool again on its next turn.
|
|
1099
1104
|
*/
|
|
1100
1105
|
private rejectHandoff;
|
|
1106
|
+
/**
|
|
1107
|
+
* First-turn deafness shields an agent-first greeting from early caller
|
|
1108
|
+
* speech where the bridge owns barge-in. A full-duplex model owns talk-over
|
|
1109
|
+
* itself, and "deaf" there only means fed silence — so the default flips
|
|
1110
|
+
* off on `turnTaking: 'model'`; an explicit setting is honored as written.
|
|
1111
|
+
*/
|
|
1112
|
+
private firstTurnDeafness;
|
|
1113
|
+
/**
|
|
1114
|
+
* The provider session behind every open response is being closed (handoff
|
|
1115
|
+
* or reconnect): their tail marks may never come. Finalize them now, or
|
|
1116
|
+
* `isPlaybackActive()` stays true for the rest of the call and everything
|
|
1117
|
+
* gated on it — the hangup grace, REST transfers, guard rotation — wedges
|
|
1118
|
+
* (field, Sept 2026). Whatever Twilio still holds plays out; no
|
|
1119
|
+
* `playback.finished` is claimed for audio nobody confirmed.
|
|
1120
|
+
*/
|
|
1121
|
+
private abandonOpenPlayback;
|
|
1122
|
+
/**
|
|
1123
|
+
* On a full-duplex provider the backend's transfer lands while the voice is
|
|
1124
|
+
* still speaking the sentence that announces it, and the handoff is a
|
|
1125
|
+
* close-and-reopen: performing it at once cuts that sentence, and the
|
|
1126
|
+
* in-flight deltas kill the handoff hold before it starts (field, Sept
|
|
1127
|
+
* 2026). Hold the transfer until the utterance has played out plus one
|
|
1128
|
+
* sentence gap — a new utterance cancels the grace — capped so a voice that
|
|
1129
|
+
* never stops still hands off.
|
|
1130
|
+
*/
|
|
1131
|
+
private deferHandoff;
|
|
1132
|
+
private maybeRunPendingHandoff;
|
|
1133
|
+
private clearPendingHandoff;
|
|
1101
1134
|
private performHandoff;
|
|
1102
1135
|
/**
|
|
1103
1136
|
* Burst-write a stored μ-law greeting straight onto the Twilio socket —
|
|
@@ -1276,6 +1309,18 @@ declare class PlaybackTracker {
|
|
|
1276
1309
|
playedMs: number;
|
|
1277
1310
|
itemId?: string;
|
|
1278
1311
|
}>;
|
|
1312
|
+
/**
|
|
1313
|
+
* The provider session behind every open response is gone (closed for a
|
|
1314
|
+
* handoff or reconnect), so their remaining marks may never come home:
|
|
1315
|
+
* finalize them at their current playedMs so `isPlaybackActive()` cannot
|
|
1316
|
+
* stay true for the rest of the call. Late echoes classify as flushed.
|
|
1317
|
+
* Returns the abandoned responses.
|
|
1318
|
+
*/
|
|
1319
|
+
abandonOpen(): Array<{
|
|
1320
|
+
responseId: string;
|
|
1321
|
+
playedMs: number;
|
|
1322
|
+
itemId?: string;
|
|
1323
|
+
}>;
|
|
1279
1324
|
/**
|
|
1280
1325
|
* Best-estimate of what the caller has heard of `responseId` right now:
|
|
1281
1326
|
* last confirmed mark plus wall-clock elapsed since, clamped to the total.
|
package/dist/index.mjs
CHANGED
|
@@ -531,6 +531,28 @@ var PlaybackTracker = class {
|
|
|
531
531
|
return interrupted;
|
|
532
532
|
}
|
|
533
533
|
/**
|
|
534
|
+
* The provider session behind every open response is gone (closed for a
|
|
535
|
+
* handoff or reconnect), so their remaining marks may never come home:
|
|
536
|
+
* finalize them at their current playedMs so `isPlaybackActive()` cannot
|
|
537
|
+
* stay true for the rest of the call. Late echoes classify as flushed.
|
|
538
|
+
* Returns the abandoned responses.
|
|
539
|
+
*/
|
|
540
|
+
abandonOpen() {
|
|
541
|
+
const abandoned = [];
|
|
542
|
+
for (const track of [...this.responses.values()]) {
|
|
543
|
+
if (track.finished) continue;
|
|
544
|
+
track.flushed = true;
|
|
545
|
+
track.finished = true;
|
|
546
|
+
abandoned.push({
|
|
547
|
+
responseId: track.responseId,
|
|
548
|
+
playedMs: track.playedMs,
|
|
549
|
+
itemId: track.itemId
|
|
550
|
+
});
|
|
551
|
+
this.maybeForget(track);
|
|
552
|
+
}
|
|
553
|
+
return abandoned;
|
|
554
|
+
}
|
|
555
|
+
/**
|
|
534
556
|
* Best-estimate of what the caller has heard of `responseId` right now:
|
|
535
557
|
* last confirmed mark plus wall-clock elapsed since, clamped to the total.
|
|
536
558
|
*/
|
|
@@ -1095,6 +1117,8 @@ var CallSession = class extends TypedEmitter {
|
|
|
1095
1117
|
reconnecting = false;
|
|
1096
1118
|
pendingHangup = null;
|
|
1097
1119
|
pendingTransfer = null;
|
|
1120
|
+
/** Model-owned turn-taking: a transfer that landed mid-utterance, held until the sentence plays out. */
|
|
1121
|
+
pendingHandoff = null;
|
|
1098
1122
|
endedReason = null;
|
|
1099
1123
|
hangupReason = "agent-hangup";
|
|
1100
1124
|
middlewares;
|
|
@@ -1487,6 +1511,11 @@ var CallSession = class extends TypedEmitter {
|
|
|
1487
1511
|
this.pendingHangup.grace = null;
|
|
1488
1512
|
}
|
|
1489
1513
|
}
|
|
1514
|
+
if (this.pendingHandoff?.grace) {
|
|
1515
|
+
clearTimeout(this.pendingHandoff.grace);
|
|
1516
|
+
this.timers.delete(this.pendingHandoff.grace);
|
|
1517
|
+
this.pendingHandoff.grace = null;
|
|
1518
|
+
}
|
|
1490
1519
|
this.noteHangupProgress();
|
|
1491
1520
|
this.clearIdleTimer();
|
|
1492
1521
|
this.interruptions.onResponseStarted(responseId);
|
|
@@ -1507,6 +1536,7 @@ var CallSession = class extends TypedEmitter {
|
|
|
1507
1536
|
this.flushToolQueue();
|
|
1508
1537
|
this.executePendingTransfer();
|
|
1509
1538
|
this.maybeCompleteHangup();
|
|
1539
|
+
this.maybeRunPendingHandoff();
|
|
1510
1540
|
}
|
|
1511
1541
|
});
|
|
1512
1542
|
provider.on("agentTranscript", ({ responseId, text }) => {
|
|
@@ -1574,7 +1604,7 @@ var CallSession = class extends TypedEmitter {
|
|
|
1574
1604
|
const proposal = this.noiseVad.onInboundFrame(payload, excluded);
|
|
1575
1605
|
if (proposal) this.handleVadProposal(proposal);
|
|
1576
1606
|
}
|
|
1577
|
-
if (this.pregreeting !== null && !this.pregreeting.played || this.
|
|
1607
|
+
if (this.pregreeting !== null && !this.pregreeting.played || this.firstTurnDeafness() && !this.firstTurnDone || this.deps.options.deafness.muteDuringToolExecution && this.runningTools.size > 0 || this.deps.options.deafness.muteWhileAgentSpeaking && this.tracker.isPlaybackActive() || this.interruptions.isSuspended) {
|
|
1578
1608
|
if (!this.modelOwnsTurnTaking()) return;
|
|
1579
1609
|
payload = silenceLike(payload);
|
|
1580
1610
|
}
|
|
@@ -1669,6 +1699,7 @@ var CallSession = class extends TypedEmitter {
|
|
|
1669
1699
|
this.flushToolQueue();
|
|
1670
1700
|
this.executePendingTransfer();
|
|
1671
1701
|
this.maybeCompleteHangup();
|
|
1702
|
+
this.maybeRunPendingHandoff();
|
|
1672
1703
|
if (this.blockedUserTurn === "committed") {
|
|
1673
1704
|
this.blockedUserTurn = "idle";
|
|
1674
1705
|
if (!hadQueuedToolResults && this.runningTools.size === 0) this.provider?.createResponse();
|
|
@@ -1858,6 +1889,10 @@ var CallSession = class extends TypedEmitter {
|
|
|
1858
1889
|
result: { handoffTo: target.id },
|
|
1859
1890
|
durationMs: Date.now() - started
|
|
1860
1891
|
});
|
|
1892
|
+
if (this.modelOwnsTurnTaking() && (this.generating || this.tracker.isPlaybackActive())) {
|
|
1893
|
+
this.deferHandoff(target, directive.reason);
|
|
1894
|
+
return;
|
|
1895
|
+
}
|
|
1861
1896
|
await this.performHandoff(target, directive.reason);
|
|
1862
1897
|
return;
|
|
1863
1898
|
}
|
|
@@ -2158,7 +2193,7 @@ var CallSession = class extends TypedEmitter {
|
|
|
2158
2193
|
pending.grace = null;
|
|
2159
2194
|
if (this.generating || this.tracker.isPlaybackActive()) return;
|
|
2160
2195
|
this.completeHangup();
|
|
2161
|
-
},
|
|
2196
|
+
}, SENTENCE_GRACE_MS);
|
|
2162
2197
|
timer.unref?.();
|
|
2163
2198
|
this.timers.add(timer);
|
|
2164
2199
|
this.pendingHangup.grace = timer;
|
|
@@ -2225,6 +2260,7 @@ var CallSession = class extends TypedEmitter {
|
|
|
2225
2260
|
if (this.stateValue !== "active") return;
|
|
2226
2261
|
const provider = this.provider;
|
|
2227
2262
|
if (!provider) return;
|
|
2263
|
+
this.abandonOpenPlayback();
|
|
2228
2264
|
try {
|
|
2229
2265
|
await provider.connect(this.buildProviderInit());
|
|
2230
2266
|
} catch (error) {
|
|
@@ -2273,6 +2309,89 @@ var CallSession = class extends TypedEmitter {
|
|
|
2273
2309
|
cause: "no-caller-turn"
|
|
2274
2310
|
});
|
|
2275
2311
|
}
|
|
2312
|
+
/**
|
|
2313
|
+
* First-turn deafness shields an agent-first greeting from early caller
|
|
2314
|
+
* speech where the bridge owns barge-in. A full-duplex model owns talk-over
|
|
2315
|
+
* itself, and "deaf" there only means fed silence — so the default flips
|
|
2316
|
+
* off on `turnTaking: 'model'`; an explicit setting is honored as written.
|
|
2317
|
+
*/
|
|
2318
|
+
firstTurnDeafness() {
|
|
2319
|
+
return this.deps.options.deafness.ignoreUserAudioUntilFirstTurnDone ?? !this.modelOwnsTurnTaking();
|
|
2320
|
+
}
|
|
2321
|
+
/**
|
|
2322
|
+
* The provider session behind every open response is being closed (handoff
|
|
2323
|
+
* or reconnect): their tail marks may never come. Finalize them now, or
|
|
2324
|
+
* `isPlaybackActive()` stays true for the rest of the call and everything
|
|
2325
|
+
* gated on it — the hangup grace, REST transfers, guard rotation — wedges
|
|
2326
|
+
* (field, Sept 2026). Whatever Twilio still holds plays out; no
|
|
2327
|
+
* `playback.finished` is claimed for audio nobody confirmed.
|
|
2328
|
+
*/
|
|
2329
|
+
abandonOpenPlayback() {
|
|
2330
|
+
if (this.tracker.abandonOpen().length === 0) return;
|
|
2331
|
+
this.interruptions.onPlaybackEnded();
|
|
2332
|
+
this.maybeCompleteHangup();
|
|
2333
|
+
this.maybeRunPendingHandoff();
|
|
2334
|
+
}
|
|
2335
|
+
/**
|
|
2336
|
+
* On a full-duplex provider the backend's transfer lands while the voice is
|
|
2337
|
+
* still speaking the sentence that announces it, and the handoff is a
|
|
2338
|
+
* close-and-reopen: performing it at once cuts that sentence, and the
|
|
2339
|
+
* in-flight deltas kill the handoff hold before it starts (field, Sept
|
|
2340
|
+
* 2026). Hold the transfer until the utterance has played out plus one
|
|
2341
|
+
* sentence gap — a new utterance cancels the grace — capped so a voice that
|
|
2342
|
+
* never stops still hands off.
|
|
2343
|
+
*/
|
|
2344
|
+
deferHandoff(target, reason) {
|
|
2345
|
+
if (this.pendingHandoff) {
|
|
2346
|
+
this.pendingHandoff.target = target;
|
|
2347
|
+
this.pendingHandoff.reason = reason;
|
|
2348
|
+
return;
|
|
2349
|
+
}
|
|
2350
|
+
const cap = setTimeout(() => {
|
|
2351
|
+
this.timers.delete(cap);
|
|
2352
|
+
const pending = this.pendingHandoff;
|
|
2353
|
+
if (!pending) return;
|
|
2354
|
+
this.log.warn("deferred handoff capped — the voice kept talking; handing off now");
|
|
2355
|
+
this.clearPendingHandoff();
|
|
2356
|
+
this.performHandoff(pending.target, pending.reason);
|
|
2357
|
+
}, HANDOFF_DEFER_MAX_MS);
|
|
2358
|
+
cap.unref?.();
|
|
2359
|
+
this.timers.add(cap);
|
|
2360
|
+
this.pendingHandoff = {
|
|
2361
|
+
target,
|
|
2362
|
+
reason,
|
|
2363
|
+
grace: null,
|
|
2364
|
+
cap
|
|
2365
|
+
};
|
|
2366
|
+
this.maybeRunPendingHandoff();
|
|
2367
|
+
}
|
|
2368
|
+
maybeRunPendingHandoff() {
|
|
2369
|
+
const pending = this.pendingHandoff;
|
|
2370
|
+
if (!pending || pending.grace) return;
|
|
2371
|
+
if (this.generating || this.tracker.isPlaybackActive()) return;
|
|
2372
|
+
const grace = setTimeout(() => {
|
|
2373
|
+
this.timers.delete(grace);
|
|
2374
|
+
const current = this.pendingHandoff;
|
|
2375
|
+
if (!current) return;
|
|
2376
|
+
current.grace = null;
|
|
2377
|
+
if (this.generating || this.tracker.isPlaybackActive()) return;
|
|
2378
|
+
this.clearPendingHandoff();
|
|
2379
|
+
this.performHandoff(current.target, current.reason);
|
|
2380
|
+
}, SENTENCE_GRACE_MS);
|
|
2381
|
+
grace.unref?.();
|
|
2382
|
+
this.timers.add(grace);
|
|
2383
|
+
pending.grace = grace;
|
|
2384
|
+
}
|
|
2385
|
+
clearPendingHandoff() {
|
|
2386
|
+
const pending = this.pendingHandoff;
|
|
2387
|
+
if (!pending) return;
|
|
2388
|
+
this.pendingHandoff = null;
|
|
2389
|
+
for (const timer of [pending.grace, pending.cap]) {
|
|
2390
|
+
if (!timer) continue;
|
|
2391
|
+
clearTimeout(timer);
|
|
2392
|
+
this.timers.delete(timer);
|
|
2393
|
+
}
|
|
2394
|
+
}
|
|
2276
2395
|
async performHandoff(target, reason) {
|
|
2277
2396
|
if (this.stateValue !== "active" || !this.provider) return;
|
|
2278
2397
|
if (target.id === this.activeAgentValue.id) return;
|
|
@@ -2305,6 +2424,7 @@ var CallSession = class extends TypedEmitter {
|
|
|
2305
2424
|
startDelayMs: hold.startDelayMs ?? 300
|
|
2306
2425
|
});
|
|
2307
2426
|
try {
|
|
2427
|
+
this.abandonOpenPlayback();
|
|
2308
2428
|
await this.provider.close();
|
|
2309
2429
|
await this.provider.connect({
|
|
2310
2430
|
...this.buildProviderInit(),
|
|
@@ -2505,7 +2625,10 @@ var CallSession = class extends TypedEmitter {
|
|
|
2505
2625
|
};
|
|
2506
2626
|
const PREGREETING_MARK = "pre:greeting";
|
|
2507
2627
|
/** Longer than the speech gate's quiet window plus the mark round trip (0.8 s + ~0.3 s). */
|
|
2508
|
-
|
|
2628
|
+
/** Model-owned turn-taking: the gate can split one sentence at a pause — wait one gap for the next. */
|
|
2629
|
+
const SENTENCE_GRACE_MS = 1500;
|
|
2630
|
+
/** A deferred handoff runs no later than this after the transfer landed, even mid-utterance. */
|
|
2631
|
+
const HANDOFF_DEFER_MAX_MS = 5e3;
|
|
2509
2632
|
/** 400ms per frame — matches production burst-write implementations. */
|
|
2510
2633
|
const PREGREETING_CHUNK_BYTES = 3200;
|
|
2511
2634
|
function toError(value) {
|
|
@@ -2532,7 +2655,6 @@ const DEFAULT_SESSION_OPTIONS = {
|
|
|
2532
2655
|
greeting: { mode: "agent-initiates" },
|
|
2533
2656
|
interruptions: { enabled: true },
|
|
2534
2657
|
deafness: {
|
|
2535
|
-
ignoreUserAudioUntilFirstTurnDone: true,
|
|
2536
2658
|
muteDuringToolExecution: true,
|
|
2537
2659
|
muteWhileAgentSpeaking: false
|
|
2538
2660
|
},
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "realtime-voice-agents",
|
|
3
|
-
"version": "2.5.
|
|
3
|
+
"version": "2.5.2",
|
|
4
4
|
"description": "Provider-agnostic bridge between Twilio Media Streams and realtime speech-to-speech AI APIs (OpenAI Realtime, OpenAI GPT-Live full-duplex, xAI Grok Voice, Gemini Live). Multi-agent handoffs, Zod tools with execution strategies, mark-based playback tracking, interruption guards, and hold audio — for Node.js voice agents over the phone.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"twilio",
|