realtime-voice-agents 2.5.0 → 2.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -119,7 +119,9 @@ One `SessionOptions` surface configures all four; where a provider can't honor a
119
119
  - **Tools never pause the voice.** Results are delivered the moment they are ready regardless of `toolResultDelivery`; an interruption does not cancel a running tool, and its result still reaches the backend. Results are relayed in the model's own words — use exact wording only through the voice prompt.
120
120
  - **Greetings, nudges and goodbyes** (`greeting.instructions`, `idle.prompts`, `finish_call`) are delivered as `session.commentary.append` — the append that reliably produces speech on demand. Keypad entries and deferred results are `session.thinking.append`; runtime instructions are `session.instructions.append`. Each append is capped at 500 tokens (long texts are split).
121
121
  - **Immutable session.** Instructions, voice and audio format cannot change after start, so handoffs and reconnects open a fresh session and seed the attributed transcript through `session.input` (≤ 128 messages) — the anti-loop replay is preserved. Sessions expire after 120 minutes; an expiry reconnects the same way.
122
- - **Deafness feeds silence.** The model's session clock runs on input audio, so `deafness` options replace caller audio with silence instead of dropping frames.
122
+ - **Transfers wait for the sentence.** A handoff here is a close-and-reopen, and the backend's transfer lands while the voice is still announcing it — the bridge holds the handoff until that utterance has played out (plus one sentence gap, capped at 5 s), so nothing is cut mid-word and `session.handoffHold` audio covers the reopen. Prompt the voice to *delegate first, announce after*: a transfer or tool the voice announces without delegating never happens.
123
+ - **Deafness feeds silence.** The model's session clock runs on input audio, so `deafness` options replace caller audio with silence instead of dropping frames. `ignoreUserAudioUntilFirstTurnDone` therefore defaults to **off** here — the model handles talk-over itself; set it explicitly to keep the greeting deaf.
124
+ - **Real-time stream, 200 ms of cushion.** The voice arrives at exactly real-time pace, so Twilio's buffer never runs ahead of playout and every delivery hiccup between OpenAI, your server and Twilio would be an audible gap (Realtime generates faster than real time, so it never has this problem). The provider holds the first 200 ms of each utterance — the last idle delta included, so soft onsets are not clipped — then streams through. `gptLive({ playoutLeadMs })` tunes it, `0` disables; the cost is that much latency on each turn's first word.
123
125
  - **Billing is per second** of session (plus backend tokens). `session.usage.audioSeconds` carries the running total; backend token usage is summed from `response.completed`. The provider sends `session.close` on teardown and waits for `session.closed`, so a hung-up call never keeps billing.
124
126
 
125
127
  ## Provider fallbacks
@@ -341,7 +343,7 @@ session: {
341
343
  greeting: { mode: 'agent-initiates' }, // 'user-initiates' to wait
342
344
  interruptions: { enabled: true },
343
345
  deafness: {
344
- ignoreUserAudioUntilFirstTurnDone: true, // auto-false with greeting.mode 'user-initiates' (caller must be heard to start)
346
+ ignoreUserAudioUntilFirstTurnDone: undefined, // auto: true, except false with greeting.mode 'user-initiates' and on full-duplex providers (GPT-Live)
345
347
  muteDuringToolExecution: true,
346
348
  muteWhileAgentSpeaking: false, // half-duplex: deaf while agent audio plays (caller speech is lost, not queued)
347
349
  },
@@ -355,6 +357,7 @@ session: {
355
357
  toolResultDelivery: 'afterPlayback', // or 'immediate'
356
358
  toolBackgroundAudio: undefined, // default hold audio for tools
357
359
  handoffVoicePolicy: 'keep', // or 'reconnect' to switch voices
360
+ handoffHold: undefined, // { spec: 'ringing', ... }: hold audio over a reconnect-style handoff (Gemini, GPT-Live)
358
361
  context: {}, // seed session KV for tools/instructions
359
362
  }
360
363
  ```
package/dist/gpt-live.cjs CHANGED
@@ -310,6 +310,8 @@ const NON_RETRIABLE_CLOSE_CODES = /* @__PURE__ */ new Set([
310
310
  const APPEND_MAX_CHARS = 1200;
311
311
  const TRANSCRIPT_IDLE_EXTRA_MS = 300;
312
312
  const GATE_STALL_EXTRA_MS = 500;
313
+ const DEFAULT_PLAYOUT_LEAD_MS = 200;
314
+ const MULAW_BYTES_PER_MS = 8;
313
315
  const ACK_TYPES = {
314
316
  "session.instructions.append": "session.instructions.appended",
315
317
  "session.thinking.append": "session.thinking.appended",
@@ -346,6 +348,10 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
346
348
  expiresAt = null;
347
349
  gate;
348
350
  gateStallTimer = null;
351
+ /** Deltas of the current utterance held until the playout lead has accumulated. */
352
+ lead = null;
353
+ /** The last idle delta: becomes the utterance's pre-roll when the gate opens. */
354
+ preRoll = null;
349
355
  utteranceCounter = 0;
350
356
  currentUtteranceId = null;
351
357
  lastUtteranceId = null;
@@ -492,6 +498,8 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
492
498
  this.inputTranscript.dispose();
493
499
  this.outputTranscript.dispose();
494
500
  this.clearGateStall();
501
+ this.lead = null;
502
+ this.preRoll = null;
495
503
  if (!ws$2 || ws$2.readyState === ws.default.CLOSED) return;
496
504
  await new Promise((resolve) => {
497
505
  const timer = setTimeout(() => finish(), this.config.closeTimeoutMs ?? 3e3);
@@ -685,14 +693,63 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
685
693
  for (const gateEvent of events) if (gateEvent.type === "open") {
686
694
  this.beginUtterance();
687
695
  forwardId = this.currentUtteranceId;
696
+ this.armPlayoutLead();
688
697
  } else close = gateEvent;
689
- if (forwardId) this.emit("audio", {
690
- base64Mulaw: delta,
691
- responseId: forwardId
692
- });
698
+ if (forwardId) this.forwardAudio(delta, bytes.length / MULAW_BYTES_PER_MS, forwardId);
699
+ else this.preRoll = {
700
+ delta,
701
+ ms: bytes.length / MULAW_BYTES_PER_MS
702
+ };
693
703
  if (close) this.endUtterance();
694
704
  else if (this.gate.isOpen) this.armGateStall();
695
705
  }
706
+ /**
707
+ * The stream is real-time paced, so Twilio's buffer never runs ahead of
708
+ * playout and every delivery hiccup between the model, this server and
709
+ * Twilio is an audible gap (field, Sept 2026: "slightly choppy"). Hold the
710
+ * first `playoutLeadMs` of each utterance — the last idle delta included —
711
+ * then flush and stream through: Twilio keeps that much cushion for the
712
+ * rest of the utterance, at the cost of that much latency on its first word.
713
+ */
714
+ armPlayoutLead() {
715
+ if (this.playoutLeadMs() <= 0) {
716
+ this.preRoll = null;
717
+ return;
718
+ }
719
+ this.lead = {
720
+ pending: [],
721
+ pendingMs: 0
722
+ };
723
+ if (this.preRoll) {
724
+ this.lead.pending.push(this.preRoll.delta);
725
+ this.lead.pendingMs += this.preRoll.ms;
726
+ this.preRoll = null;
727
+ }
728
+ }
729
+ forwardAudio(delta, ms, responseId) {
730
+ if (!this.lead) {
731
+ this.emit("audio", {
732
+ base64Mulaw: delta,
733
+ responseId
734
+ });
735
+ return;
736
+ }
737
+ this.lead.pending.push(delta);
738
+ this.lead.pendingMs += ms;
739
+ if (this.lead.pendingMs >= this.playoutLeadMs()) this.flushPlayoutLead(responseId);
740
+ }
741
+ flushPlayoutLead(responseId) {
742
+ const lead = this.lead;
743
+ if (!lead) return;
744
+ this.lead = null;
745
+ for (const delta of lead.pending) this.emit("audio", {
746
+ base64Mulaw: delta,
747
+ responseId
748
+ });
749
+ }
750
+ playoutLeadMs() {
751
+ return this.config.playoutLeadMs ?? 200;
752
+ }
696
753
  handleBackendEvent(inner, delegationId) {
697
754
  const type = inner.type ?? "";
698
755
  if (type === "response.output_item.done" && inner.item?.type === "function_call") {
@@ -729,6 +786,7 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
729
786
  endUtterance() {
730
787
  this.clearGateStall();
731
788
  const id = this.currentUtteranceId;
789
+ if (id) this.flushPlayoutLead(id);
732
790
  this.currentUtteranceId = null;
733
791
  if (id) this.emit("responseDone", { responseId: id });
734
792
  }
@@ -757,6 +815,8 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
757
815
  resetUtteranceState() {
758
816
  this.clearGateStall();
759
817
  this.gate.close();
818
+ this.lead = null;
819
+ this.preRoll = null;
760
820
  this.currentUtteranceId = null;
761
821
  this.lastUtteranceId = null;
762
822
  this.seenCalls.clear();
@@ -923,6 +983,7 @@ function gptLive(options = {}) {
923
983
  store: options.store,
924
984
  extraSessionOptions: options.sessionOptions,
925
985
  connectTimeoutMs: options.connectTimeoutMs,
986
+ playoutLeadMs: options.playoutLeadMs,
926
987
  speechGate: options.speechGate,
927
988
  transcriptGapMs: options.transcriptGapMs
928
989
  };
@@ -938,6 +999,7 @@ function gptLive(options = {}) {
938
999
  //#endregion
939
1000
  exports.DEFAULT_GATE_QUIET_MS = DEFAULT_GATE_QUIET_MS;
940
1001
  exports.DEFAULT_GATE_THRESHOLD_RMS = DEFAULT_GATE_THRESHOLD_RMS;
1002
+ exports.DEFAULT_PLAYOUT_LEAD_MS = DEFAULT_PLAYOUT_LEAD_MS;
941
1003
  exports.GPT_LIVE_AUDIO_FORMAT = GPT_LIVE_AUDIO_FORMAT;
942
1004
  exports.GPT_LIVE_DEFAULT_BACKEND_MODEL = GPT_LIVE_DEFAULT_BACKEND_MODEL;
943
1005
  exports.GPT_LIVE_DEFAULT_BASE_URL = GPT_LIVE_DEFAULT_BASE_URL;
@@ -126,6 +126,14 @@ interface GptLiveProviderConfig {
126
126
  connectTimeoutMs?: number;
127
127
  /** How long `close()` waits for `session.closed` (final usage) before dropping the socket. Default 3000. */
128
128
  closeTimeoutMs?: number;
129
+ /**
130
+ * Audio held back at the start of each utterance before forwarding begins,
131
+ * so Twilio keeps that much cushion against delivery jitter (the stream is
132
+ * real-time paced — without it any hiccup is an audible gap). The last idle
133
+ * delta before the onset is included, so soft onsets are not clipped.
134
+ * Default 200; 0 forwards every delta as it arrives.
135
+ */
136
+ playoutLeadMs?: number;
129
137
  /** Speech gate tuning (utterance boundaries synthesized from the audio). */
130
138
  speechGate?: SpeechGateOptions;
131
139
  /** Session-timeline gap that splits transcript fragments into turns. Default 800. */
@@ -137,6 +145,7 @@ interface GptLiveProviderConfig {
137
145
  ackTimeoutMs?: number;
138
146
  providerName?: string;
139
147
  }
148
+ declare const DEFAULT_PLAYOUT_LEAD_MS = 200;
140
149
  declare class GptLiveProvider extends BaseRealtimeProvider {
141
150
  readonly name: string;
142
151
  readonly capabilities: ProviderCapabilities;
@@ -152,6 +161,10 @@ declare class GptLiveProvider extends BaseRealtimeProvider {
152
161
  expiresAt: number | null;
153
162
  private readonly gate;
154
163
  private gateStallTimer;
164
+ /** Deltas of the current utterance held until the playout lead has accumulated. */
165
+ private lead;
166
+ /** The last idle delta: becomes the utterance's pre-roll when the gate opens. */
167
+ private preRoll;
155
168
  private utteranceCounter;
156
169
  private currentUtteranceId;
157
170
  private lastUtteranceId;
@@ -192,6 +205,18 @@ declare class GptLiveProvider extends BaseRealtimeProvider {
192
205
  updateSession(patch: Partial<ProviderSessionInit>, options?: SessionUpdateOptions): Promise<boolean>;
193
206
  private handleEvent;
194
207
  private handleOutputAudio;
208
+ /**
209
+ * The stream is real-time paced, so Twilio's buffer never runs ahead of
210
+ * playout and every delivery hiccup between the model, this server and
211
+ * Twilio is an audible gap (field, Sept 2026: "slightly choppy"). Hold the
212
+ * first `playoutLeadMs` of each utterance — the last idle delta included —
213
+ * then flush and stream through: Twilio keeps that much cushion for the
214
+ * rest of the utterance, at the cost of that much latency on its first word.
215
+ */
216
+ private armPlayoutLead;
217
+ private forwardAudio;
218
+ private flushPlayoutLead;
219
+ private playoutLeadMs;
195
220
  private handleBackendEvent;
196
221
  private beginUtterance;
197
222
  private endUtterance;
@@ -244,6 +269,14 @@ interface GptLiveOptions {
244
269
  /** Provider-native `session.start` fields, deep-merged last. The schema is strict: an unknown field rejects the session. */
245
270
  sessionOptions?: Record<string, unknown>;
246
271
  connectTimeoutMs?: number;
272
+ /**
273
+ * Cushion held at the start of each utterance before audio is forwarded to
274
+ * Twilio. The model streams at exactly real-time pace, so without it any
275
+ * delivery hiccup is an audible gap; with it Twilio stays that far ahead of
276
+ * playout. Costs the same amount of latency on each turn's first word.
277
+ * Default 200; 0 forwards every delta as it arrives.
278
+ */
279
+ playoutLeadMs?: number;
247
280
  /** Speech gate tuning — utterance boundaries synthesized from the continuous output stream. */
248
281
  speechGate?: SpeechGateOptions;
249
282
  /** Session-timeline gap that splits transcript fragments into turns. Default 800. */
@@ -258,4 +291,4 @@ interface GptLiveOptions {
258
291
  */
259
292
  declare function gptLive(options?: GptLiveOptions): ProviderFactory;
260
293
  //#endregion
261
- export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, type GptLiveDelegationOptions, GptLiveOptions, GptLiveProvider, type GptLiveProviderConfig, type GptLiveSessionConfig, GptLiveVoice, SpeechGate, type SpeechGateOptions, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
294
+ export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, DEFAULT_PLAYOUT_LEAD_MS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, type GptLiveDelegationOptions, GptLiveOptions, GptLiveProvider, type GptLiveProviderConfig, type GptLiveSessionConfig, GptLiveVoice, SpeechGate, type SpeechGateOptions, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
@@ -126,6 +126,14 @@ interface GptLiveProviderConfig {
126
126
  connectTimeoutMs?: number;
127
127
  /** How long `close()` waits for `session.closed` (final usage) before dropping the socket. Default 3000. */
128
128
  closeTimeoutMs?: number;
129
+ /**
130
+ * Audio held back at the start of each utterance before forwarding begins,
131
+ * so Twilio keeps that much cushion against delivery jitter (the stream is
132
+ * real-time paced — without it any hiccup is an audible gap). The last idle
133
+ * delta before the onset is included, so soft onsets are not clipped.
134
+ * Default 200; 0 forwards every delta as it arrives.
135
+ */
136
+ playoutLeadMs?: number;
129
137
  /** Speech gate tuning (utterance boundaries synthesized from the audio). */
130
138
  speechGate?: SpeechGateOptions;
131
139
  /** Session-timeline gap that splits transcript fragments into turns. Default 800. */
@@ -137,6 +145,7 @@ interface GptLiveProviderConfig {
137
145
  ackTimeoutMs?: number;
138
146
  providerName?: string;
139
147
  }
148
+ declare const DEFAULT_PLAYOUT_LEAD_MS = 200;
140
149
  declare class GptLiveProvider extends BaseRealtimeProvider {
141
150
  readonly name: string;
142
151
  readonly capabilities: ProviderCapabilities;
@@ -152,6 +161,10 @@ declare class GptLiveProvider extends BaseRealtimeProvider {
152
161
  expiresAt: number | null;
153
162
  private readonly gate;
154
163
  private gateStallTimer;
164
+ /** Deltas of the current utterance held until the playout lead has accumulated. */
165
+ private lead;
166
+ /** The last idle delta: becomes the utterance's pre-roll when the gate opens. */
167
+ private preRoll;
155
168
  private utteranceCounter;
156
169
  private currentUtteranceId;
157
170
  private lastUtteranceId;
@@ -192,6 +205,18 @@ declare class GptLiveProvider extends BaseRealtimeProvider {
192
205
  updateSession(patch: Partial<ProviderSessionInit>, options?: SessionUpdateOptions): Promise<boolean>;
193
206
  private handleEvent;
194
207
  private handleOutputAudio;
208
+ /**
209
+ * The stream is real-time paced, so Twilio's buffer never runs ahead of
210
+ * playout and every delivery hiccup between the model, this server and
211
+ * Twilio is an audible gap (field, Sept 2026: "slightly choppy"). Hold the
212
+ * first `playoutLeadMs` of each utterance — the last idle delta included —
213
+ * then flush and stream through: Twilio keeps that much cushion for the
214
+ * rest of the utterance, at the cost of that much latency on its first word.
215
+ */
216
+ private armPlayoutLead;
217
+ private forwardAudio;
218
+ private flushPlayoutLead;
219
+ private playoutLeadMs;
195
220
  private handleBackendEvent;
196
221
  private beginUtterance;
197
222
  private endUtterance;
@@ -244,6 +269,14 @@ interface GptLiveOptions {
244
269
  /** Provider-native `session.start` fields, deep-merged last. The schema is strict: an unknown field rejects the session. */
245
270
  sessionOptions?: Record<string, unknown>;
246
271
  connectTimeoutMs?: number;
272
+ /**
273
+ * Cushion held at the start of each utterance before audio is forwarded to
274
+ * Twilio. The model streams at exactly real-time pace, so without it any
275
+ * delivery hiccup is an audible gap; with it Twilio stays that far ahead of
276
+ * playout. Costs the same amount of latency on each turn's first word.
277
+ * Default 200; 0 forwards every delta as it arrives.
278
+ */
279
+ playoutLeadMs?: number;
247
280
  /** Speech gate tuning — utterance boundaries synthesized from the continuous output stream. */
248
281
  speechGate?: SpeechGateOptions;
249
282
  /** Session-timeline gap that splits transcript fragments into turns. Default 800. */
@@ -258,4 +291,4 @@ interface GptLiveOptions {
258
291
  */
259
292
  declare function gptLive(options?: GptLiveOptions): ProviderFactory;
260
293
  //#endregion
261
- export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, type GptLiveDelegationOptions, GptLiveOptions, GptLiveProvider, type GptLiveProviderConfig, type GptLiveSessionConfig, GptLiveVoice, SpeechGate, type SpeechGateOptions, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
294
+ export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, DEFAULT_PLAYOUT_LEAD_MS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, type GptLiveDelegationOptions, GptLiveOptions, GptLiveProvider, type GptLiveProviderConfig, type GptLiveSessionConfig, GptLiveVoice, SpeechGate, type SpeechGateOptions, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
package/dist/gpt-live.mjs CHANGED
@@ -307,6 +307,8 @@ const NON_RETRIABLE_CLOSE_CODES = /* @__PURE__ */ new Set([
307
307
  const APPEND_MAX_CHARS = 1200;
308
308
  const TRANSCRIPT_IDLE_EXTRA_MS = 300;
309
309
  const GATE_STALL_EXTRA_MS = 500;
310
+ const DEFAULT_PLAYOUT_LEAD_MS = 200;
311
+ const MULAW_BYTES_PER_MS = 8;
310
312
  const ACK_TYPES = {
311
313
  "session.instructions.append": "session.instructions.appended",
312
314
  "session.thinking.append": "session.thinking.appended",
@@ -343,6 +345,10 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
343
345
  expiresAt = null;
344
346
  gate;
345
347
  gateStallTimer = null;
348
+ /** Deltas of the current utterance held until the playout lead has accumulated. */
349
+ lead = null;
350
+ /** The last idle delta: becomes the utterance's pre-roll when the gate opens. */
351
+ preRoll = null;
346
352
  utteranceCounter = 0;
347
353
  currentUtteranceId = null;
348
354
  lastUtteranceId = null;
@@ -489,6 +495,8 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
489
495
  this.inputTranscript.dispose();
490
496
  this.outputTranscript.dispose();
491
497
  this.clearGateStall();
498
+ this.lead = null;
499
+ this.preRoll = null;
492
500
  if (!ws || ws.readyState === WebSocket$1.CLOSED) return;
493
501
  await new Promise((resolve) => {
494
502
  const timer = setTimeout(() => finish(), this.config.closeTimeoutMs ?? 3e3);
@@ -682,14 +690,63 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
682
690
  for (const gateEvent of events) if (gateEvent.type === "open") {
683
691
  this.beginUtterance();
684
692
  forwardId = this.currentUtteranceId;
693
+ this.armPlayoutLead();
685
694
  } else close = gateEvent;
686
- if (forwardId) this.emit("audio", {
687
- base64Mulaw: delta,
688
- responseId: forwardId
689
- });
695
+ if (forwardId) this.forwardAudio(delta, bytes.length / MULAW_BYTES_PER_MS, forwardId);
696
+ else this.preRoll = {
697
+ delta,
698
+ ms: bytes.length / MULAW_BYTES_PER_MS
699
+ };
690
700
  if (close) this.endUtterance();
691
701
  else if (this.gate.isOpen) this.armGateStall();
692
702
  }
703
+ /**
704
+ * The stream is real-time paced, so Twilio's buffer never runs ahead of
705
+ * playout and every delivery hiccup between the model, this server and
706
+ * Twilio is an audible gap (field, Sept 2026: "slightly choppy"). Hold the
707
+ * first `playoutLeadMs` of each utterance — the last idle delta included —
708
+ * then flush and stream through: Twilio keeps that much cushion for the
709
+ * rest of the utterance, at the cost of that much latency on its first word.
710
+ */
711
+ armPlayoutLead() {
712
+ if (this.playoutLeadMs() <= 0) {
713
+ this.preRoll = null;
714
+ return;
715
+ }
716
+ this.lead = {
717
+ pending: [],
718
+ pendingMs: 0
719
+ };
720
+ if (this.preRoll) {
721
+ this.lead.pending.push(this.preRoll.delta);
722
+ this.lead.pendingMs += this.preRoll.ms;
723
+ this.preRoll = null;
724
+ }
725
+ }
726
+ forwardAudio(delta, ms, responseId) {
727
+ if (!this.lead) {
728
+ this.emit("audio", {
729
+ base64Mulaw: delta,
730
+ responseId
731
+ });
732
+ return;
733
+ }
734
+ this.lead.pending.push(delta);
735
+ this.lead.pendingMs += ms;
736
+ if (this.lead.pendingMs >= this.playoutLeadMs()) this.flushPlayoutLead(responseId);
737
+ }
738
+ flushPlayoutLead(responseId) {
739
+ const lead = this.lead;
740
+ if (!lead) return;
741
+ this.lead = null;
742
+ for (const delta of lead.pending) this.emit("audio", {
743
+ base64Mulaw: delta,
744
+ responseId
745
+ });
746
+ }
747
+ playoutLeadMs() {
748
+ return this.config.playoutLeadMs ?? 200;
749
+ }
693
750
  handleBackendEvent(inner, delegationId) {
694
751
  const type = inner.type ?? "";
695
752
  if (type === "response.output_item.done" && inner.item?.type === "function_call") {
@@ -726,6 +783,7 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
726
783
  endUtterance() {
727
784
  this.clearGateStall();
728
785
  const id = this.currentUtteranceId;
786
+ if (id) this.flushPlayoutLead(id);
729
787
  this.currentUtteranceId = null;
730
788
  if (id) this.emit("responseDone", { responseId: id });
731
789
  }
@@ -754,6 +812,8 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
754
812
  resetUtteranceState() {
755
813
  this.clearGateStall();
756
814
  this.gate.close();
815
+ this.lead = null;
816
+ this.preRoll = null;
757
817
  this.currentUtteranceId = null;
758
818
  this.lastUtteranceId = null;
759
819
  this.seenCalls.clear();
@@ -920,6 +980,7 @@ function gptLive(options = {}) {
920
980
  store: options.store,
921
981
  extraSessionOptions: options.sessionOptions,
922
982
  connectTimeoutMs: options.connectTimeoutMs,
983
+ playoutLeadMs: options.playoutLeadMs,
923
984
  speechGate: options.speechGate,
924
985
  transcriptGapMs: options.transcriptGapMs
925
986
  };
@@ -933,4 +994,4 @@ function gptLive(options = {}) {
933
994
  };
934
995
  }
935
996
  //#endregion
936
- export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, GptLiveProvider, SpeechGate, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
997
+ export { DEFAULT_GATE_QUIET_MS, DEFAULT_GATE_THRESHOLD_RMS, DEFAULT_PLAYOUT_LEAD_MS, GPT_LIVE_AUDIO_FORMAT, GPT_LIVE_DEFAULT_BACKEND_MODEL, GPT_LIVE_DEFAULT_BASE_URL, GPT_LIVE_DEFAULT_MODEL, GPT_LIVE_DEFAULT_VOICE, GPT_LIVE_KEY_ENV_VARS, GPT_LIVE_VOICES, GptLiveProvider, SpeechGate, buildDelegation, buildHistoryItems, buildSessionStart, gptLive, splitForAppend };
package/dist/index.cjs CHANGED
@@ -535,6 +535,28 @@ var PlaybackTracker = class {
535
535
  return interrupted;
536
536
  }
537
537
  /**
538
+ * The provider session behind every open response is gone (closed for a
539
+ * handoff or reconnect), so their remaining marks may never come home:
540
+ * finalize them at their current playedMs so `isPlaybackActive()` cannot
541
+ * stay true for the rest of the call. Late echoes classify as flushed.
542
+ * Returns the abandoned responses.
543
+ */
544
+ abandonOpen() {
545
+ const abandoned = [];
546
+ for (const track of [...this.responses.values()]) {
547
+ if (track.finished) continue;
548
+ track.flushed = true;
549
+ track.finished = true;
550
+ abandoned.push({
551
+ responseId: track.responseId,
552
+ playedMs: track.playedMs,
553
+ itemId: track.itemId
554
+ });
555
+ this.maybeForget(track);
556
+ }
557
+ return abandoned;
558
+ }
559
+ /**
538
560
  * Best-estimate of what the caller has heard of `responseId` right now:
539
561
  * last confirmed mark plus wall-clock elapsed since, clamped to the total.
540
562
  */
@@ -1099,6 +1121,8 @@ var CallSession = class extends require_events.TypedEmitter {
1099
1121
  reconnecting = false;
1100
1122
  pendingHangup = null;
1101
1123
  pendingTransfer = null;
1124
+ /** Model-owned turn-taking: a transfer that landed mid-utterance, held until the sentence plays out. */
1125
+ pendingHandoff = null;
1102
1126
  endedReason = null;
1103
1127
  hangupReason = "agent-hangup";
1104
1128
  middlewares;
@@ -1491,6 +1515,11 @@ var CallSession = class extends require_events.TypedEmitter {
1491
1515
  this.pendingHangup.grace = null;
1492
1516
  }
1493
1517
  }
1518
+ if (this.pendingHandoff?.grace) {
1519
+ clearTimeout(this.pendingHandoff.grace);
1520
+ this.timers.delete(this.pendingHandoff.grace);
1521
+ this.pendingHandoff.grace = null;
1522
+ }
1494
1523
  this.noteHangupProgress();
1495
1524
  this.clearIdleTimer();
1496
1525
  this.interruptions.onResponseStarted(responseId);
@@ -1511,6 +1540,7 @@ var CallSession = class extends require_events.TypedEmitter {
1511
1540
  this.flushToolQueue();
1512
1541
  this.executePendingTransfer();
1513
1542
  this.maybeCompleteHangup();
1543
+ this.maybeRunPendingHandoff();
1514
1544
  }
1515
1545
  });
1516
1546
  provider.on("agentTranscript", ({ responseId, text }) => {
@@ -1578,7 +1608,7 @@ var CallSession = class extends require_events.TypedEmitter {
1578
1608
  const proposal = this.noiseVad.onInboundFrame(payload, excluded);
1579
1609
  if (proposal) this.handleVadProposal(proposal);
1580
1610
  }
1581
- if (this.pregreeting !== null && !this.pregreeting.played || this.deps.options.deafness.ignoreUserAudioUntilFirstTurnDone && !this.firstTurnDone || this.deps.options.deafness.muteDuringToolExecution && this.runningTools.size > 0 || this.deps.options.deafness.muteWhileAgentSpeaking && this.tracker.isPlaybackActive() || this.interruptions.isSuspended) {
1611
+ if (this.pregreeting !== null && !this.pregreeting.played || this.firstTurnDeafness() && !this.firstTurnDone || this.deps.options.deafness.muteDuringToolExecution && this.runningTools.size > 0 || this.deps.options.deafness.muteWhileAgentSpeaking && this.tracker.isPlaybackActive() || this.interruptions.isSuspended) {
1582
1612
  if (!this.modelOwnsTurnTaking()) return;
1583
1613
  payload = silenceLike(payload);
1584
1614
  }
@@ -1673,6 +1703,7 @@ var CallSession = class extends require_events.TypedEmitter {
1673
1703
  this.flushToolQueue();
1674
1704
  this.executePendingTransfer();
1675
1705
  this.maybeCompleteHangup();
1706
+ this.maybeRunPendingHandoff();
1676
1707
  if (this.blockedUserTurn === "committed") {
1677
1708
  this.blockedUserTurn = "idle";
1678
1709
  if (!hadQueuedToolResults && this.runningTools.size === 0) this.provider?.createResponse();
@@ -1862,6 +1893,10 @@ var CallSession = class extends require_events.TypedEmitter {
1862
1893
  result: { handoffTo: target.id },
1863
1894
  durationMs: Date.now() - started
1864
1895
  });
1896
+ if (this.modelOwnsTurnTaking() && (this.generating || this.tracker.isPlaybackActive())) {
1897
+ this.deferHandoff(target, directive.reason);
1898
+ return;
1899
+ }
1865
1900
  await this.performHandoff(target, directive.reason);
1866
1901
  return;
1867
1902
  }
@@ -2162,7 +2197,7 @@ var CallSession = class extends require_events.TypedEmitter {
2162
2197
  pending.grace = null;
2163
2198
  if (this.generating || this.tracker.isPlaybackActive()) return;
2164
2199
  this.completeHangup();
2165
- }, HANGUP_SENTENCE_GRACE_MS);
2200
+ }, SENTENCE_GRACE_MS);
2166
2201
  timer.unref?.();
2167
2202
  this.timers.add(timer);
2168
2203
  this.pendingHangup.grace = timer;
@@ -2229,6 +2264,7 @@ var CallSession = class extends require_events.TypedEmitter {
2229
2264
  if (this.stateValue !== "active") return;
2230
2265
  const provider = this.provider;
2231
2266
  if (!provider) return;
2267
+ this.abandonOpenPlayback();
2232
2268
  try {
2233
2269
  await provider.connect(this.buildProviderInit());
2234
2270
  } catch (error) {
@@ -2277,6 +2313,89 @@ var CallSession = class extends require_events.TypedEmitter {
2277
2313
  cause: "no-caller-turn"
2278
2314
  });
2279
2315
  }
2316
+ /**
2317
+ * First-turn deafness shields an agent-first greeting from early caller
2318
+ * speech where the bridge owns barge-in. A full-duplex model owns talk-over
2319
+ * itself, and "deaf" there only means fed silence — so the default flips
2320
+ * off on `turnTaking: 'model'`; an explicit setting is honored as written.
2321
+ */
2322
+ firstTurnDeafness() {
2323
+ return this.deps.options.deafness.ignoreUserAudioUntilFirstTurnDone ?? !this.modelOwnsTurnTaking();
2324
+ }
2325
+ /**
2326
+ * The provider session behind every open response is being closed (handoff
2327
+ * or reconnect): their tail marks may never come. Finalize them now, or
2328
+ * `isPlaybackActive()` stays true for the rest of the call and everything
2329
+ * gated on it — the hangup grace, REST transfers, guard rotation — wedges
2330
+ * (field, Sept 2026). Whatever Twilio still holds plays out; no
2331
+ * `playback.finished` is claimed for audio nobody confirmed.
2332
+ */
2333
+ abandonOpenPlayback() {
2334
+ if (this.tracker.abandonOpen().length === 0) return;
2335
+ this.interruptions.onPlaybackEnded();
2336
+ this.maybeCompleteHangup();
2337
+ this.maybeRunPendingHandoff();
2338
+ }
2339
+ /**
2340
+ * On a full-duplex provider the backend's transfer lands while the voice is
2341
+ * still speaking the sentence that announces it, and the handoff is a
2342
+ * close-and-reopen: performing it at once cuts that sentence, and the
2343
+ * in-flight deltas kill the handoff hold before it starts (field, Sept
2344
+ * 2026). Hold the transfer until the utterance has played out plus one
2345
+ * sentence gap — a new utterance cancels the grace — capped so a voice that
2346
+ * never stops still hands off.
2347
+ */
2348
+ deferHandoff(target, reason) {
2349
+ if (this.pendingHandoff) {
2350
+ this.pendingHandoff.target = target;
2351
+ this.pendingHandoff.reason = reason;
2352
+ return;
2353
+ }
2354
+ const cap = setTimeout(() => {
2355
+ this.timers.delete(cap);
2356
+ const pending = this.pendingHandoff;
2357
+ if (!pending) return;
2358
+ this.log.warn("deferred handoff capped — the voice kept talking; handing off now");
2359
+ this.clearPendingHandoff();
2360
+ this.performHandoff(pending.target, pending.reason);
2361
+ }, HANDOFF_DEFER_MAX_MS);
2362
+ cap.unref?.();
2363
+ this.timers.add(cap);
2364
+ this.pendingHandoff = {
2365
+ target,
2366
+ reason,
2367
+ grace: null,
2368
+ cap
2369
+ };
2370
+ this.maybeRunPendingHandoff();
2371
+ }
2372
+ maybeRunPendingHandoff() {
2373
+ const pending = this.pendingHandoff;
2374
+ if (!pending || pending.grace) return;
2375
+ if (this.generating || this.tracker.isPlaybackActive()) return;
2376
+ const grace = setTimeout(() => {
2377
+ this.timers.delete(grace);
2378
+ const current = this.pendingHandoff;
2379
+ if (!current) return;
2380
+ current.grace = null;
2381
+ if (this.generating || this.tracker.isPlaybackActive()) return;
2382
+ this.clearPendingHandoff();
2383
+ this.performHandoff(current.target, current.reason);
2384
+ }, SENTENCE_GRACE_MS);
2385
+ grace.unref?.();
2386
+ this.timers.add(grace);
2387
+ pending.grace = grace;
2388
+ }
2389
+ clearPendingHandoff() {
2390
+ const pending = this.pendingHandoff;
2391
+ if (!pending) return;
2392
+ this.pendingHandoff = null;
2393
+ for (const timer of [pending.grace, pending.cap]) {
2394
+ if (!timer) continue;
2395
+ clearTimeout(timer);
2396
+ this.timers.delete(timer);
2397
+ }
2398
+ }
2280
2399
  async performHandoff(target, reason) {
2281
2400
  if (this.stateValue !== "active" || !this.provider) return;
2282
2401
  if (target.id === this.activeAgentValue.id) return;
@@ -2309,6 +2428,7 @@ var CallSession = class extends require_events.TypedEmitter {
2309
2428
  startDelayMs: hold.startDelayMs ?? 300
2310
2429
  });
2311
2430
  try {
2431
+ this.abandonOpenPlayback();
2312
2432
  await this.provider.close();
2313
2433
  await this.provider.connect({
2314
2434
  ...this.buildProviderInit(),
@@ -2509,7 +2629,10 @@ var CallSession = class extends require_events.TypedEmitter {
2509
2629
  };
2510
2630
  const PREGREETING_MARK = "pre:greeting";
2511
2631
  /** Longer than the speech gate's quiet window plus the mark round trip (0.8 s + ~0.3 s). */
2512
- const HANGUP_SENTENCE_GRACE_MS = 1500;
2632
+ /** Model-owned turn-taking: the gate can split one sentence at a pause — wait one gap for the next. */
2633
+ const SENTENCE_GRACE_MS = 1500;
2634
+ /** A deferred handoff runs no later than this after the transfer landed, even mid-utterance. */
2635
+ const HANDOFF_DEFER_MAX_MS = 5e3;
2513
2636
  /** 400ms per frame — matches production burst-write implementations. */
2514
2637
  const PREGREETING_CHUNK_BYTES = 3200;
2515
2638
  function toError(value) {
@@ -2536,7 +2659,6 @@ const DEFAULT_SESSION_OPTIONS = {
2536
2659
  greeting: { mode: "agent-initiates" },
2537
2660
  interruptions: { enabled: true },
2538
2661
  deafness: {
2539
- ignoreUserAudioUntilFirstTurnDone: true,
2540
2662
  muteDuringToolExecution: true,
2541
2663
  muteWhileAgentSpeaking: false
2542
2664
  },
package/dist/index.d.cts CHANGED
@@ -521,11 +521,14 @@ interface GreetingOptions {
521
521
  interface DeafnessOptions {
522
522
  /**
523
523
  * Drop caller audio until the agent's first turn finishes playing.
524
- * Protects the greeting from noisy pickups. Default true — except with
525
- * `greeting.mode: 'user-initiates'`, where the caller must be heard to
526
- * start the call at all, so the default flips to false. An explicit true
527
- * is honored even there, but deafens the call until something else
528
- * (an idle nudge, a tool) produces the agent's first turn.
524
+ * Protects the greeting from noisy pickups on providers where the bridge
525
+ * owns barge-in. Unset = automatic: true there, false with
526
+ * `greeting.mode: 'user-initiates'` (the caller must be heard to start the
527
+ * call at all) and on full-duplex providers (`turnTaking: 'model'`), where
528
+ * the model owns talk-over and "deaf" only means fed silence. An explicit
529
+ * value is honored as written — an explicit true with user-initiates
530
+ * deafens the call until something else (an idle nudge, a tool) produces
531
+ * the agent's first turn.
529
532
  */
530
533
  ignoreUserAudioUntilFirstTurnDone?: boolean;
531
534
  /** Drop caller audio while a foreground tool is running. Default true. */
@@ -914,6 +917,8 @@ declare class CallSession extends TypedEmitter<SessionEventMap> {
914
917
  private reconnecting;
915
918
  private pendingHangup;
916
919
  private pendingTransfer;
920
+ /** Model-owned turn-taking: a transfer that landed mid-utterance, held until the sentence plays out. */
921
+ private pendingHandoff;
917
922
  private endedReason;
918
923
  private hangupReason;
919
924
  private readonly middlewares;
@@ -1098,6 +1103,34 @@ declare class CallSession extends TypedEmitter<SessionEventMap> {
1098
1103
  * calls the same transfer tool again on its next turn.
1099
1104
  */
1100
1105
  private rejectHandoff;
1106
+ /**
1107
+ * First-turn deafness shields an agent-first greeting from early caller
1108
+ * speech where the bridge owns barge-in. A full-duplex model owns talk-over
1109
+ * itself, and "deaf" there only means fed silence — so the default flips
1110
+ * off on `turnTaking: 'model'`; an explicit setting is honored as written.
1111
+ */
1112
+ private firstTurnDeafness;
1113
+ /**
1114
+ * The provider session behind every open response is being closed (handoff
1115
+ * or reconnect): their tail marks may never come. Finalize them now, or
1116
+ * `isPlaybackActive()` stays true for the rest of the call and everything
1117
+ * gated on it — the hangup grace, REST transfers, guard rotation — wedges
1118
+ * (field, Sept 2026). Whatever Twilio still holds plays out; no
1119
+ * `playback.finished` is claimed for audio nobody confirmed.
1120
+ */
1121
+ private abandonOpenPlayback;
1122
+ /**
1123
+ * On a full-duplex provider the backend's transfer lands while the voice is
1124
+ * still speaking the sentence that announces it, and the handoff is a
1125
+ * close-and-reopen: performing it at once cuts that sentence, and the
1126
+ * in-flight deltas kill the handoff hold before it starts (field, Sept
1127
+ * 2026). Hold the transfer until the utterance has played out plus one
1128
+ * sentence gap — a new utterance cancels the grace — capped so a voice that
1129
+ * never stops still hands off.
1130
+ */
1131
+ private deferHandoff;
1132
+ private maybeRunPendingHandoff;
1133
+ private clearPendingHandoff;
1101
1134
  private performHandoff;
1102
1135
  /**
1103
1136
  * Burst-write a stored μ-law greeting straight onto the Twilio socket —
@@ -1276,6 +1309,18 @@ declare class PlaybackTracker {
1276
1309
  playedMs: number;
1277
1310
  itemId?: string;
1278
1311
  }>;
1312
+ /**
1313
+ * The provider session behind every open response is gone (closed for a
1314
+ * handoff or reconnect), so their remaining marks may never come home:
1315
+ * finalize them at their current playedMs so `isPlaybackActive()` cannot
1316
+ * stay true for the rest of the call. Late echoes classify as flushed.
1317
+ * Returns the abandoned responses.
1318
+ */
1319
+ abandonOpen(): Array<{
1320
+ responseId: string;
1321
+ playedMs: number;
1322
+ itemId?: string;
1323
+ }>;
1279
1324
  /**
1280
1325
  * Best-estimate of what the caller has heard of `responseId` right now:
1281
1326
  * last confirmed mark plus wall-clock elapsed since, clamped to the total.
package/dist/index.d.mts CHANGED
@@ -521,11 +521,14 @@ interface GreetingOptions {
521
521
  interface DeafnessOptions {
522
522
  /**
523
523
  * Drop caller audio until the agent's first turn finishes playing.
524
- * Protects the greeting from noisy pickups. Default true — except with
525
- * `greeting.mode: 'user-initiates'`, where the caller must be heard to
526
- * start the call at all, so the default flips to false. An explicit true
527
- * is honored even there, but deafens the call until something else
528
- * (an idle nudge, a tool) produces the agent's first turn.
524
+ * Protects the greeting from noisy pickups on providers where the bridge
525
+ * owns barge-in. Unset = automatic: true there, false with
526
+ * `greeting.mode: 'user-initiates'` (the caller must be heard to start the
527
+ * call at all) and on full-duplex providers (`turnTaking: 'model'`), where
528
+ * the model owns talk-over and "deaf" only means fed silence. An explicit
529
+ * value is honored as written — an explicit true with user-initiates
530
+ * deafens the call until something else (an idle nudge, a tool) produces
531
+ * the agent's first turn.
529
532
  */
530
533
  ignoreUserAudioUntilFirstTurnDone?: boolean;
531
534
  /** Drop caller audio while a foreground tool is running. Default true. */
@@ -914,6 +917,8 @@ declare class CallSession extends TypedEmitter<SessionEventMap> {
914
917
  private reconnecting;
915
918
  private pendingHangup;
916
919
  private pendingTransfer;
920
+ /** Model-owned turn-taking: a transfer that landed mid-utterance, held until the sentence plays out. */
921
+ private pendingHandoff;
917
922
  private endedReason;
918
923
  private hangupReason;
919
924
  private readonly middlewares;
@@ -1098,6 +1103,34 @@ declare class CallSession extends TypedEmitter<SessionEventMap> {
1098
1103
  * calls the same transfer tool again on its next turn.
1099
1104
  */
1100
1105
  private rejectHandoff;
1106
+ /**
1107
+ * First-turn deafness shields an agent-first greeting from early caller
1108
+ * speech where the bridge owns barge-in. A full-duplex model owns talk-over
1109
+ * itself, and "deaf" there only means fed silence — so the default flips
1110
+ * off on `turnTaking: 'model'`; an explicit setting is honored as written.
1111
+ */
1112
+ private firstTurnDeafness;
1113
+ /**
1114
+ * The provider session behind every open response is being closed (handoff
1115
+ * or reconnect): their tail marks may never come. Finalize them now, or
1116
+ * `isPlaybackActive()` stays true for the rest of the call and everything
1117
+ * gated on it — the hangup grace, REST transfers, guard rotation — wedges
1118
+ * (field, Sept 2026). Whatever Twilio still holds plays out; no
1119
+ * `playback.finished` is claimed for audio nobody confirmed.
1120
+ */
1121
+ private abandonOpenPlayback;
1122
+ /**
1123
+ * On a full-duplex provider the backend's transfer lands while the voice is
1124
+ * still speaking the sentence that announces it, and the handoff is a
1125
+ * close-and-reopen: performing it at once cuts that sentence, and the
1126
+ * in-flight deltas kill the handoff hold before it starts (field, Sept
1127
+ * 2026). Hold the transfer until the utterance has played out plus one
1128
+ * sentence gap — a new utterance cancels the grace — capped so a voice that
1129
+ * never stops still hands off.
1130
+ */
1131
+ private deferHandoff;
1132
+ private maybeRunPendingHandoff;
1133
+ private clearPendingHandoff;
1101
1134
  private performHandoff;
1102
1135
  /**
1103
1136
  * Burst-write a stored μ-law greeting straight onto the Twilio socket —
@@ -1276,6 +1309,18 @@ declare class PlaybackTracker {
1276
1309
  playedMs: number;
1277
1310
  itemId?: string;
1278
1311
  }>;
1312
+ /**
1313
+ * The provider session behind every open response is gone (closed for a
1314
+ * handoff or reconnect), so their remaining marks may never come home:
1315
+ * finalize them at their current playedMs so `isPlaybackActive()` cannot
1316
+ * stay true for the rest of the call. Late echoes classify as flushed.
1317
+ * Returns the abandoned responses.
1318
+ */
1319
+ abandonOpen(): Array<{
1320
+ responseId: string;
1321
+ playedMs: number;
1322
+ itemId?: string;
1323
+ }>;
1279
1324
  /**
1280
1325
  * Best-estimate of what the caller has heard of `responseId` right now:
1281
1326
  * last confirmed mark plus wall-clock elapsed since, clamped to the total.
package/dist/index.mjs CHANGED
@@ -531,6 +531,28 @@ var PlaybackTracker = class {
531
531
  return interrupted;
532
532
  }
533
533
  /**
534
+ * The provider session behind every open response is gone (closed for a
535
+ * handoff or reconnect), so their remaining marks may never come home:
536
+ * finalize them at their current playedMs so `isPlaybackActive()` cannot
537
+ * stay true for the rest of the call. Late echoes classify as flushed.
538
+ * Returns the abandoned responses.
539
+ */
540
+ abandonOpen() {
541
+ const abandoned = [];
542
+ for (const track of [...this.responses.values()]) {
543
+ if (track.finished) continue;
544
+ track.flushed = true;
545
+ track.finished = true;
546
+ abandoned.push({
547
+ responseId: track.responseId,
548
+ playedMs: track.playedMs,
549
+ itemId: track.itemId
550
+ });
551
+ this.maybeForget(track);
552
+ }
553
+ return abandoned;
554
+ }
555
+ /**
534
556
  * Best-estimate of what the caller has heard of `responseId` right now:
535
557
  * last confirmed mark plus wall-clock elapsed since, clamped to the total.
536
558
  */
@@ -1095,6 +1117,8 @@ var CallSession = class extends TypedEmitter {
1095
1117
  reconnecting = false;
1096
1118
  pendingHangup = null;
1097
1119
  pendingTransfer = null;
1120
+ /** Model-owned turn-taking: a transfer that landed mid-utterance, held until the sentence plays out. */
1121
+ pendingHandoff = null;
1098
1122
  endedReason = null;
1099
1123
  hangupReason = "agent-hangup";
1100
1124
  middlewares;
@@ -1487,6 +1511,11 @@ var CallSession = class extends TypedEmitter {
1487
1511
  this.pendingHangup.grace = null;
1488
1512
  }
1489
1513
  }
1514
+ if (this.pendingHandoff?.grace) {
1515
+ clearTimeout(this.pendingHandoff.grace);
1516
+ this.timers.delete(this.pendingHandoff.grace);
1517
+ this.pendingHandoff.grace = null;
1518
+ }
1490
1519
  this.noteHangupProgress();
1491
1520
  this.clearIdleTimer();
1492
1521
  this.interruptions.onResponseStarted(responseId);
@@ -1507,6 +1536,7 @@ var CallSession = class extends TypedEmitter {
1507
1536
  this.flushToolQueue();
1508
1537
  this.executePendingTransfer();
1509
1538
  this.maybeCompleteHangup();
1539
+ this.maybeRunPendingHandoff();
1510
1540
  }
1511
1541
  });
1512
1542
  provider.on("agentTranscript", ({ responseId, text }) => {
@@ -1574,7 +1604,7 @@ var CallSession = class extends TypedEmitter {
1574
1604
  const proposal = this.noiseVad.onInboundFrame(payload, excluded);
1575
1605
  if (proposal) this.handleVadProposal(proposal);
1576
1606
  }
1577
- if (this.pregreeting !== null && !this.pregreeting.played || this.deps.options.deafness.ignoreUserAudioUntilFirstTurnDone && !this.firstTurnDone || this.deps.options.deafness.muteDuringToolExecution && this.runningTools.size > 0 || this.deps.options.deafness.muteWhileAgentSpeaking && this.tracker.isPlaybackActive() || this.interruptions.isSuspended) {
1607
+ if (this.pregreeting !== null && !this.pregreeting.played || this.firstTurnDeafness() && !this.firstTurnDone || this.deps.options.deafness.muteDuringToolExecution && this.runningTools.size > 0 || this.deps.options.deafness.muteWhileAgentSpeaking && this.tracker.isPlaybackActive() || this.interruptions.isSuspended) {
1578
1608
  if (!this.modelOwnsTurnTaking()) return;
1579
1609
  payload = silenceLike(payload);
1580
1610
  }
@@ -1669,6 +1699,7 @@ var CallSession = class extends TypedEmitter {
1669
1699
  this.flushToolQueue();
1670
1700
  this.executePendingTransfer();
1671
1701
  this.maybeCompleteHangup();
1702
+ this.maybeRunPendingHandoff();
1672
1703
  if (this.blockedUserTurn === "committed") {
1673
1704
  this.blockedUserTurn = "idle";
1674
1705
  if (!hadQueuedToolResults && this.runningTools.size === 0) this.provider?.createResponse();
@@ -1858,6 +1889,10 @@ var CallSession = class extends TypedEmitter {
1858
1889
  result: { handoffTo: target.id },
1859
1890
  durationMs: Date.now() - started
1860
1891
  });
1892
+ if (this.modelOwnsTurnTaking() && (this.generating || this.tracker.isPlaybackActive())) {
1893
+ this.deferHandoff(target, directive.reason);
1894
+ return;
1895
+ }
1861
1896
  await this.performHandoff(target, directive.reason);
1862
1897
  return;
1863
1898
  }
@@ -2158,7 +2193,7 @@ var CallSession = class extends TypedEmitter {
2158
2193
  pending.grace = null;
2159
2194
  if (this.generating || this.tracker.isPlaybackActive()) return;
2160
2195
  this.completeHangup();
2161
- }, HANGUP_SENTENCE_GRACE_MS);
2196
+ }, SENTENCE_GRACE_MS);
2162
2197
  timer.unref?.();
2163
2198
  this.timers.add(timer);
2164
2199
  this.pendingHangup.grace = timer;
@@ -2225,6 +2260,7 @@ var CallSession = class extends TypedEmitter {
2225
2260
  if (this.stateValue !== "active") return;
2226
2261
  const provider = this.provider;
2227
2262
  if (!provider) return;
2263
+ this.abandonOpenPlayback();
2228
2264
  try {
2229
2265
  await provider.connect(this.buildProviderInit());
2230
2266
  } catch (error) {
@@ -2273,6 +2309,89 @@ var CallSession = class extends TypedEmitter {
2273
2309
  cause: "no-caller-turn"
2274
2310
  });
2275
2311
  }
2312
+ /**
2313
+ * First-turn deafness shields an agent-first greeting from early caller
2314
+ * speech where the bridge owns barge-in. A full-duplex model owns talk-over
2315
+ * itself, and "deaf" there only means fed silence — so the default flips
2316
+ * off on `turnTaking: 'model'`; an explicit setting is honored as written.
2317
+ */
2318
+ firstTurnDeafness() {
2319
+ return this.deps.options.deafness.ignoreUserAudioUntilFirstTurnDone ?? !this.modelOwnsTurnTaking();
2320
+ }
2321
+ /**
2322
+ * The provider session behind every open response is being closed (handoff
2323
+ * or reconnect): their tail marks may never come. Finalize them now, or
2324
+ * `isPlaybackActive()` stays true for the rest of the call and everything
2325
+ * gated on it — the hangup grace, REST transfers, guard rotation — wedges
2326
+ * (field, Sept 2026). Whatever Twilio still holds plays out; no
2327
+ * `playback.finished` is claimed for audio nobody confirmed.
2328
+ */
2329
+ abandonOpenPlayback() {
2330
+ if (this.tracker.abandonOpen().length === 0) return;
2331
+ this.interruptions.onPlaybackEnded();
2332
+ this.maybeCompleteHangup();
2333
+ this.maybeRunPendingHandoff();
2334
+ }
2335
+ /**
2336
+ * On a full-duplex provider the backend's transfer lands while the voice is
2337
+ * still speaking the sentence that announces it, and the handoff is a
2338
+ * close-and-reopen: performing it at once cuts that sentence, and the
2339
+ * in-flight deltas kill the handoff hold before it starts (field, Sept
2340
+ * 2026). Hold the transfer until the utterance has played out plus one
2341
+ * sentence gap — a new utterance cancels the grace — capped so a voice that
2342
+ * never stops still hands off.
2343
+ */
2344
+ deferHandoff(target, reason) {
2345
+ if (this.pendingHandoff) {
2346
+ this.pendingHandoff.target = target;
2347
+ this.pendingHandoff.reason = reason;
2348
+ return;
2349
+ }
2350
+ const cap = setTimeout(() => {
2351
+ this.timers.delete(cap);
2352
+ const pending = this.pendingHandoff;
2353
+ if (!pending) return;
2354
+ this.log.warn("deferred handoff capped — the voice kept talking; handing off now");
2355
+ this.clearPendingHandoff();
2356
+ this.performHandoff(pending.target, pending.reason);
2357
+ }, HANDOFF_DEFER_MAX_MS);
2358
+ cap.unref?.();
2359
+ this.timers.add(cap);
2360
+ this.pendingHandoff = {
2361
+ target,
2362
+ reason,
2363
+ grace: null,
2364
+ cap
2365
+ };
2366
+ this.maybeRunPendingHandoff();
2367
+ }
2368
+ maybeRunPendingHandoff() {
2369
+ const pending = this.pendingHandoff;
2370
+ if (!pending || pending.grace) return;
2371
+ if (this.generating || this.tracker.isPlaybackActive()) return;
2372
+ const grace = setTimeout(() => {
2373
+ this.timers.delete(grace);
2374
+ const current = this.pendingHandoff;
2375
+ if (!current) return;
2376
+ current.grace = null;
2377
+ if (this.generating || this.tracker.isPlaybackActive()) return;
2378
+ this.clearPendingHandoff();
2379
+ this.performHandoff(current.target, current.reason);
2380
+ }, SENTENCE_GRACE_MS);
2381
+ grace.unref?.();
2382
+ this.timers.add(grace);
2383
+ pending.grace = grace;
2384
+ }
2385
+ clearPendingHandoff() {
2386
+ const pending = this.pendingHandoff;
2387
+ if (!pending) return;
2388
+ this.pendingHandoff = null;
2389
+ for (const timer of [pending.grace, pending.cap]) {
2390
+ if (!timer) continue;
2391
+ clearTimeout(timer);
2392
+ this.timers.delete(timer);
2393
+ }
2394
+ }
2276
2395
  async performHandoff(target, reason) {
2277
2396
  if (this.stateValue !== "active" || !this.provider) return;
2278
2397
  if (target.id === this.activeAgentValue.id) return;
@@ -2305,6 +2424,7 @@ var CallSession = class extends TypedEmitter {
2305
2424
  startDelayMs: hold.startDelayMs ?? 300
2306
2425
  });
2307
2426
  try {
2427
+ this.abandonOpenPlayback();
2308
2428
  await this.provider.close();
2309
2429
  await this.provider.connect({
2310
2430
  ...this.buildProviderInit(),
@@ -2505,7 +2625,10 @@ var CallSession = class extends TypedEmitter {
2505
2625
  };
2506
2626
  const PREGREETING_MARK = "pre:greeting";
2507
2627
  /** Longer than the speech gate's quiet window plus the mark round trip (0.8 s + ~0.3 s). */
2508
- const HANGUP_SENTENCE_GRACE_MS = 1500;
2628
+ /** Model-owned turn-taking: the gate can split one sentence at a pause — wait one gap for the next. */
2629
+ const SENTENCE_GRACE_MS = 1500;
2630
+ /** A deferred handoff runs no later than this after the transfer landed, even mid-utterance. */
2631
+ const HANDOFF_DEFER_MAX_MS = 5e3;
2509
2632
  /** 400ms per frame — matches production burst-write implementations. */
2510
2633
  const PREGREETING_CHUNK_BYTES = 3200;
2511
2634
  function toError(value) {
@@ -2532,7 +2655,6 @@ const DEFAULT_SESSION_OPTIONS = {
2532
2655
  greeting: { mode: "agent-initiates" },
2533
2656
  interruptions: { enabled: true },
2534
2657
  deafness: {
2535
- ignoreUserAudioUntilFirstTurnDone: true,
2536
2658
  muteDuringToolExecution: true,
2537
2659
  muteWhileAgentSpeaking: false
2538
2660
  },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "realtime-voice-agents",
3
- "version": "2.5.0",
3
+ "version": "2.5.2",
4
4
  "description": "Provider-agnostic bridge between Twilio Media Streams and realtime speech-to-speech AI APIs (OpenAI Realtime, OpenAI GPT-Live full-duplex, xAI Grok Voice, Gemini Live). Multi-agent handoffs, Zod tools with execution strategies, mark-based playback tracking, interruption guards, and hold audio — for Node.js voice agents over the phone.",
5
5
  "keywords": [
6
6
  "twilio",