realtime-voice-agents 2.5.2 → 2.5.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/gpt-live.cjs +13 -4
- package/dist/gpt-live.mjs +13 -4
- package/dist/index.cjs +7 -2
- package/dist/index.mjs +7 -2
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -119,7 +119,7 @@ One `SessionOptions` surface configures all four; where a provider can't honor a
|
|
|
119
119
|
- **Tools never pause the voice.** Results are delivered the moment they are ready regardless of `toolResultDelivery`; an interruption does not cancel a running tool, and its result still reaches the backend. Results are relayed in the model's own words — use exact wording only through the voice prompt.
|
|
120
120
|
- **Greetings, nudges and goodbyes** (`greeting.instructions`, `idle.prompts`, `finish_call`) are delivered as `session.commentary.append` — the append that reliably produces speech on demand. Keypad entries and deferred results are `session.thinking.append`; runtime instructions are `session.instructions.append`. Each append is capped at 500 tokens (long texts are split).
|
|
121
121
|
- **Immutable session.** Instructions, voice and audio format cannot change after start, so handoffs and reconnects open a fresh session and seed the attributed transcript through `session.input` (≤ 128 messages) — the anti-loop replay is preserved. Sessions expire after 120 minutes; an expiry reconnects the same way.
|
|
122
|
-
- **Transfers wait for the sentence.** A handoff here is a close-and-reopen, and the backend's transfer lands while the voice is still announcing it — the bridge holds the handoff until that utterance has played out (plus one sentence gap, capped at
|
|
122
|
+
- **Transfers wait for the sentence.** A handoff here is a close-and-reopen, and the backend's transfer lands while the voice is still announcing it — the bridge holds the handoff until that utterance has played out (plus one sentence gap, capped at 8 s), so nothing is cut mid-word and `session.handoffHold` audio covers the reopen. Prompt the voice to *delegate first, announce after*: a transfer or tool the voice announces without delegating never happens.
|
|
123
123
|
- **Deafness feeds silence.** The model's session clock runs on input audio, so `deafness` options replace caller audio with silence instead of dropping frames. `ignoreUserAudioUntilFirstTurnDone` therefore defaults to **off** here — the model handles talk-over itself; set it explicitly to keep the greeting deaf.
|
|
124
124
|
- **Real-time stream, 200 ms of cushion.** The voice arrives at exactly real-time pace, so Twilio's buffer never runs ahead of playout and every delivery hiccup between OpenAI, your server and Twilio would be an audible gap (Realtime generates faster than real time, so it never has this problem). The provider holds the first 200 ms of each utterance — the last idle delta included, so soft onsets are not clipped — then streams through. `gptLive({ playoutLeadMs })` tunes it, `0` disables; the cost is that much latency on each turn's first word.
|
|
125
125
|
- **Billing is per second** of session (plus backend tokens). `session.usage.audioSeconds` carries the running total; backend token usage is summed from `response.completed`. The provider sends `session.close` on teardown and waits for `session.closed`, so a hung-up call never keeps billing.
|
package/dist/gpt-live.cjs
CHANGED
|
@@ -689,18 +689,26 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
|
|
|
689
689
|
const wasOpen = this.gate.isOpen;
|
|
690
690
|
const events = this.gate.feed(bytes);
|
|
691
691
|
let forwardId = wasOpen ? this.currentUtteranceId : null;
|
|
692
|
-
let
|
|
693
|
-
for (const gateEvent of events)
|
|
692
|
+
let pendingClose = false;
|
|
693
|
+
for (const gateEvent of events) {
|
|
694
|
+
if (gateEvent.type === "close") {
|
|
695
|
+
pendingClose = true;
|
|
696
|
+
continue;
|
|
697
|
+
}
|
|
698
|
+
if (pendingClose) {
|
|
699
|
+
this.endUtterance();
|
|
700
|
+
pendingClose = false;
|
|
701
|
+
}
|
|
694
702
|
this.beginUtterance();
|
|
695
703
|
forwardId = this.currentUtteranceId;
|
|
696
704
|
this.armPlayoutLead();
|
|
697
|
-
}
|
|
705
|
+
}
|
|
698
706
|
if (forwardId) this.forwardAudio(delta, bytes.length / MULAW_BYTES_PER_MS, forwardId);
|
|
699
707
|
else this.preRoll = {
|
|
700
708
|
delta,
|
|
701
709
|
ms: bytes.length / MULAW_BYTES_PER_MS
|
|
702
710
|
};
|
|
703
|
-
if (
|
|
711
|
+
if (pendingClose) this.endUtterance();
|
|
704
712
|
else if (this.gate.isOpen) this.armGateStall();
|
|
705
713
|
}
|
|
706
714
|
/**
|
|
@@ -779,6 +787,7 @@ var GptLiveProvider = class extends require_BaseRealtimeProvider.BaseRealtimePro
|
|
|
779
787
|
});
|
|
780
788
|
}
|
|
781
789
|
beginUtterance() {
|
|
790
|
+
if (this.currentUtteranceId) this.endUtterance();
|
|
782
791
|
this.currentUtteranceId = `live_utt_${++this.utteranceCounter}`;
|
|
783
792
|
this.lastUtteranceId = this.currentUtteranceId;
|
|
784
793
|
this.emit("responseStarted", { responseId: this.currentUtteranceId });
|
package/dist/gpt-live.mjs
CHANGED
|
@@ -686,18 +686,26 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
|
|
|
686
686
|
const wasOpen = this.gate.isOpen;
|
|
687
687
|
const events = this.gate.feed(bytes);
|
|
688
688
|
let forwardId = wasOpen ? this.currentUtteranceId : null;
|
|
689
|
-
let
|
|
690
|
-
for (const gateEvent of events)
|
|
689
|
+
let pendingClose = false;
|
|
690
|
+
for (const gateEvent of events) {
|
|
691
|
+
if (gateEvent.type === "close") {
|
|
692
|
+
pendingClose = true;
|
|
693
|
+
continue;
|
|
694
|
+
}
|
|
695
|
+
if (pendingClose) {
|
|
696
|
+
this.endUtterance();
|
|
697
|
+
pendingClose = false;
|
|
698
|
+
}
|
|
691
699
|
this.beginUtterance();
|
|
692
700
|
forwardId = this.currentUtteranceId;
|
|
693
701
|
this.armPlayoutLead();
|
|
694
|
-
}
|
|
702
|
+
}
|
|
695
703
|
if (forwardId) this.forwardAudio(delta, bytes.length / MULAW_BYTES_PER_MS, forwardId);
|
|
696
704
|
else this.preRoll = {
|
|
697
705
|
delta,
|
|
698
706
|
ms: bytes.length / MULAW_BYTES_PER_MS
|
|
699
707
|
};
|
|
700
|
-
if (
|
|
708
|
+
if (pendingClose) this.endUtterance();
|
|
701
709
|
else if (this.gate.isOpen) this.armGateStall();
|
|
702
710
|
}
|
|
703
711
|
/**
|
|
@@ -776,6 +784,7 @@ var GptLiveProvider = class extends BaseRealtimeProvider {
|
|
|
776
784
|
});
|
|
777
785
|
}
|
|
778
786
|
beginUtterance() {
|
|
787
|
+
if (this.currentUtteranceId) this.endUtterance();
|
|
779
788
|
this.currentUtteranceId = `live_utt_${++this.utteranceCounter}`;
|
|
780
789
|
this.lastUtteranceId = this.currentUtteranceId;
|
|
781
790
|
this.emit("responseStarted", { responseId: this.currentUtteranceId });
|
package/dist/index.cjs
CHANGED
|
@@ -2631,8 +2631,13 @@ const PREGREETING_MARK = "pre:greeting";
|
|
|
2631
2631
|
/** Longer than the speech gate's quiet window plus the mark round trip (0.8 s + ~0.3 s). */
|
|
2632
2632
|
/** Model-owned turn-taking: the gate can split one sentence at a pause — wait one gap for the next. */
|
|
2633
2633
|
const SENTENCE_GRACE_MS = 1500;
|
|
2634
|
-
/**
|
|
2635
|
-
|
|
2634
|
+
/**
|
|
2635
|
+
* A deferred handoff runs no later than this after the transfer landed, even
|
|
2636
|
+
* mid-utterance. One announce + a stray utterance + the sentence grace lands
|
|
2637
|
+
* right at 5 s in the field (Sept 2026); the cap is for a voice that never
|
|
2638
|
+
* stops, and the caller hears the agent meanwhile, so it errs long.
|
|
2639
|
+
*/
|
|
2640
|
+
const HANDOFF_DEFER_MAX_MS = 8e3;
|
|
2636
2641
|
/** 400ms per frame — matches production burst-write implementations. */
|
|
2637
2642
|
const PREGREETING_CHUNK_BYTES = 3200;
|
|
2638
2643
|
function toError(value) {
|
package/dist/index.mjs
CHANGED
|
@@ -2627,8 +2627,13 @@ const PREGREETING_MARK = "pre:greeting";
|
|
|
2627
2627
|
/** Longer than the speech gate's quiet window plus the mark round trip (0.8 s + ~0.3 s). */
|
|
2628
2628
|
/** Model-owned turn-taking: the gate can split one sentence at a pause — wait one gap for the next. */
|
|
2629
2629
|
const SENTENCE_GRACE_MS = 1500;
|
|
2630
|
-
/**
|
|
2631
|
-
|
|
2630
|
+
/**
|
|
2631
|
+
* A deferred handoff runs no later than this after the transfer landed, even
|
|
2632
|
+
* mid-utterance. One announce + a stray utterance + the sentence grace lands
|
|
2633
|
+
* right at 5 s in the field (Sept 2026); the cap is for a voice that never
|
|
2634
|
+
* stops, and the caller hears the agent meanwhile, so it errs long.
|
|
2635
|
+
*/
|
|
2636
|
+
const HANDOFF_DEFER_MAX_MS = 8e3;
|
|
2632
2637
|
/** 400ms per frame — matches production burst-write implementations. */
|
|
2633
2638
|
const PREGREETING_CHUNK_BYTES = 3200;
|
|
2634
2639
|
function toError(value) {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "realtime-voice-agents",
|
|
3
|
-
"version": "2.5.
|
|
3
|
+
"version": "2.5.3",
|
|
4
4
|
"description": "Provider-agnostic bridge between Twilio Media Streams and realtime speech-to-speech AI APIs (OpenAI Realtime, OpenAI GPT-Live full-duplex, xAI Grok Voice, Gemini Live). Multi-agent handoffs, Zod tools with execution strategies, mark-based playback tracking, interruption guards, and hold audio — for Node.js voice agents over the phone.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"twilio",
|