osborn 0.9.114 → 0.9.116
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +26 -22
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -242,7 +242,7 @@ async function synthMp3(text) {
|
|
|
242
242
|
const dgKey = process.env.DEEPGRAM_API_KEY;
|
|
243
243
|
if (dgKey) {
|
|
244
244
|
try {
|
|
245
|
-
const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-
|
|
245
|
+
const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-asteria-en&encoding=mp3&bit_rate=48000', {
|
|
246
246
|
method: 'POST',
|
|
247
247
|
headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
|
|
248
248
|
body: JSON.stringify({ text: text.slice(0, 4000) }),
|
|
@@ -659,7 +659,7 @@ function startApiServer(workingDir, port) {
|
|
|
659
659
|
// WAV/linear16, not mp3: mp3 files carry encoder padding (leading/
|
|
660
660
|
// trailing silence + boundary click) — back-to-back sentence clips
|
|
661
661
|
// were heard as "cracking". WAV is gapless-safe and cheaper to decode.
|
|
662
|
-
const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-
|
|
662
|
+
const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-asteria-en&encoding=linear16&sample_rate=48000&container=wav', {
|
|
663
663
|
method: 'POST',
|
|
664
664
|
headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
|
|
665
665
|
body: JSON.stringify({ text }),
|
|
@@ -1984,10 +1984,15 @@ async function main() {
|
|
|
1984
1984
|
}
|
|
1985
1985
|
}
|
|
1986
1986
|
else {
|
|
1987
|
-
// speech_off — a natural silence boundary
|
|
1988
|
-
//
|
|
1989
|
-
|
|
1990
|
-
|
|
1987
|
+
// speech_off — a natural silence boundary = the speaker finished their
|
|
1988
|
+
// utterance. When the conversation latch is open (1:1 or interactive
|
|
1989
|
+
// mode), that boundary IS the bot's turn — flush ADDRESSED so it
|
|
1990
|
+
// replies at natural pauses, VAD-driven, no name required
|
|
1991
|
+
// (user directive 2026-08-01).
|
|
1992
|
+
if (!isBot && meetingTranscriptBuffer.length) {
|
|
1993
|
+
const latchOpen = Date.now() < meetingAddressedUntil || meetingSpeakers.size <= 1;
|
|
1994
|
+
flushMeetingBuffer(botId, latchOpen);
|
|
1995
|
+
}
|
|
1991
1996
|
}
|
|
1992
1997
|
});
|
|
1993
1998
|
}
|
|
@@ -2653,11 +2658,11 @@ async function main() {
|
|
|
2653
2658
|
// re-captures in the same room → feedback). Set by PipelineDirectLLM.chat()
|
|
2654
2659
|
// when the turn is a [MEETING —] chunk. Normal user turns are unaffected.
|
|
2655
2660
|
if (directLLM.suppressMeetingTTS) {
|
|
2656
|
-
//
|
|
2657
|
-
//
|
|
2658
|
-
//
|
|
2659
|
-
//
|
|
2660
|
-
if (meetingCanvasInRoom && activeMeetingBotId && Date.now() < meetingAddressedUntil) {
|
|
2661
|
+
// MEETING MODE CONDITION (user directive 2026-08-01): says route to
|
|
2662
|
+
// the NATIVE Recall sink by default — no room/Chrome relay (that path
|
|
2663
|
+
// was moderately-to-extremely choppy). The LiveKit parity relay stays
|
|
2664
|
+
// available behind OSBORN_MEETING_AUDIO=parity for future testing.
|
|
2665
|
+
if (process.env.OSBORN_MEETING_AUDIO === 'parity' && meetingCanvasInRoom && activeMeetingBotId && Date.now() < meetingAddressedUntil) {
|
|
2661
2666
|
console.log(`🔊🎼 meeting reply via NATIVE session.say (canvas relays): "${data.text.slice(0, 50)}"`);
|
|
2662
2667
|
markMeetingSpeaking(data.text);
|
|
2663
2668
|
// no return — normal say proceeds below
|
|
@@ -5205,17 +5210,16 @@ async function main() {
|
|
|
5205
5210
|
// the bot casts the meeting canvas as its camera+mic (below), so
|
|
5206
5211
|
// it can show visuals and speak into the meeting on demand.
|
|
5207
5212
|
await sendToFrontend({ type: 'meeting_joining', message: 'Osborn is joining your meeting...' });
|
|
5208
|
-
//
|
|
5209
|
-
//
|
|
5210
|
-
//
|
|
5211
|
-
//
|
|
5212
|
-
//
|
|
5213
|
-
//
|
|
5214
|
-
|
|
5215
|
-
|
|
5216
|
-
|
|
5217
|
-
|
|
5218
|
-
const castUrl = data.castUrl || process.env.OSBORN_MEETING_CAST_URL || canvasUrl;
|
|
5213
|
+
// AUDIO-FIRST DEFAULT (2026-08-01, proven live): the webpage-camera
|
|
5214
|
+
// output_media routes the bot's AUDIO through the canvas page's
|
|
5215
|
+
// Web Audio — which sits SUSPENDED in Recall's headless Chrome (no
|
|
5216
|
+
// user gesture) → the bot was inaudible all night. Default now = NO
|
|
5217
|
+
// webpage camera, so the bot's voice is the direct Recall
|
|
5218
|
+
// output_audio track (speakIntoMeeting → recall.outputAudio, ~1.8s
|
|
5219
|
+
// synth, verified audible + clear). The browser CAST is opt-in via
|
|
5220
|
+
// explicit data.castUrl / OSBORN_MEETING_CAST_URL until we split
|
|
5221
|
+
// video(webpage) from audio(output_audio) to reclaim it cleanly.
|
|
5222
|
+
const castUrl = data.castUrl || process.env.OSBORN_MEETING_CAST_URL || undefined;
|
|
5219
5223
|
const botId = await recallJoin.joinMeeting(meetingUrl, webhookBase, { castUrl });
|
|
5220
5224
|
const sessionId = currentLLM?.sessionId || currentResumeSessionId || 'default';
|
|
5221
5225
|
recallJoin.registerBot(botId, sessionId);
|