osborn 0.9.114 → 0.9.116

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/dist/index.js +26 -22
  2. package/package.json +1 -1
package/dist/index.js CHANGED
@@ -242,7 +242,7 @@ async function synthMp3(text) {
242
242
  const dgKey = process.env.DEEPGRAM_API_KEY;
243
243
  if (dgKey) {
244
244
  try {
245
- const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-thalia-en&encoding=mp3&bit_rate=48000', {
245
+ const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-asteria-en&encoding=mp3&bit_rate=48000', {
246
246
  method: 'POST',
247
247
  headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
248
248
  body: JSON.stringify({ text: text.slice(0, 4000) }),
@@ -659,7 +659,7 @@ function startApiServer(workingDir, port) {
659
659
  // WAV/linear16, not mp3: mp3 files carry encoder padding (leading/
660
660
  // trailing silence + boundary click) — back-to-back sentence clips
661
661
  // were heard as "cracking". WAV is gapless-safe and cheaper to decode.
662
- const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-thalia-en&encoding=linear16&sample_rate=48000&container=wav', {
662
+ const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-asteria-en&encoding=linear16&sample_rate=48000&container=wav', {
663
663
  method: 'POST',
664
664
  headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
665
665
  body: JSON.stringify({ text }),
@@ -1984,10 +1984,15 @@ async function main() {
1984
1984
  }
1985
1985
  }
1986
1986
  else {
1987
- // speech_off — a natural silence boundary. Flush the buffered turns now
1988
- // (chunk on conversational pauses, not an arbitrary clock), unless empty.
1989
- if (!isBot && meetingTranscriptBuffer.length)
1990
- flushMeetingBuffer(botId, false);
1987
+ // speech_off — a natural silence boundary = the speaker finished their
1988
+ // utterance. When the conversation latch is open (1:1 or interactive
1989
+ // mode), that boundary IS the bot's turn — flush ADDRESSED so it
1990
+ // replies at natural pauses, VAD-driven, no name required
1991
+ // (user directive 2026-08-01).
1992
+ if (!isBot && meetingTranscriptBuffer.length) {
1993
+ const latchOpen = Date.now() < meetingAddressedUntil || meetingSpeakers.size <= 1;
1994
+ flushMeetingBuffer(botId, latchOpen);
1995
+ }
1991
1996
  }
1992
1997
  });
1993
1998
  }
@@ -2653,11 +2658,11 @@ async function main() {
2653
2658
  // re-captures in the same room → feedback). Set by PipelineDirectLLM.chat()
2654
2659
  // when the turn is a [MEETING —] chunk. Normal user turns are unaffected.
2655
2660
  if (directLLM.suppressMeetingTTS) {
2656
- // BROWSER-PARITY PATH: canvas is in the LiveKit room → the normal
2657
- // session.say audio reaches the meeting through it. Fall through to
2658
- // the regular pipeline for addressed turns (identical audio to the
2659
- // browser experience); observer turns still stay silent.
2660
- if (meetingCanvasInRoom && activeMeetingBotId && Date.now() < meetingAddressedUntil) {
2661
+ // MEETING MODE CONDITION (user directive 2026-08-01): says route to
2662
+ // the NATIVE Recall sink by default — no room/Chrome relay (that path
2663
+ // was moderately-to-extremely choppy). The LiveKit parity relay stays
2664
+ // available behind OSBORN_MEETING_AUDIO=parity for future testing.
2665
+ if (process.env.OSBORN_MEETING_AUDIO === 'parity' && meetingCanvasInRoom && activeMeetingBotId && Date.now() < meetingAddressedUntil) {
2661
2666
  console.log(`🔊🎼 meeting reply via NATIVE session.say (canvas relays): "${data.text.slice(0, 50)}"`);
2662
2667
  markMeetingSpeaking(data.text);
2663
2668
  // no return — normal say proceeds below
@@ -5205,17 +5210,16 @@ async function main() {
5205
5210
  // the bot casts the meeting canvas as its camera+mic (below), so
5206
5211
  // it can show visuals and speak into the meeting on demand.
5207
5212
  await sendToFrontend({ type: 'meeting_joining', message: 'Osborn is joining your meeting...' });
5208
- // Cast target = the bot's camera+mic webpage (Recall output_media).
5209
- // DEFAULT is the meeting canvas (frontend /meeting-canvas), pointed
5210
- // at THIS agent's public URL so it can subscribe to /canvas-stream
5211
- // and become the bot's face (visuals) + voice (TTS it speaks). An
5212
- // explicit castUrl from the frontend or OSBORN_MEETING_CAST_URL
5213
- // overrides it (e.g. to cast a live feed or a seeded site instead).
5214
- const frontendUrl = (process.env.OSBORN_FRONTEND_URL || 'https://www.voice-native.com').replace(/\/$/, '');
5215
- const canvasUrl = /^https:\/\//.test(webhookBase)
5216
- ? `${frontendUrl}/meeting-canvas?agent=${encodeURIComponent(webhookBase)}`
5217
- : undefined;
5218
- const castUrl = data.castUrl || process.env.OSBORN_MEETING_CAST_URL || canvasUrl;
5213
+ // AUDIO-FIRST DEFAULT (2026-08-01, proven live): the webpage-camera
5214
+ // output_media routes the bot's AUDIO through the canvas page's
5215
+ // Web Audio — which sits SUSPENDED in Recall's headless Chrome (no
5216
+ // user gesture) → the bot was inaudible all night. Default now = NO
5217
+ // webpage camera, so the bot's voice is the direct Recall
5218
+ // output_audio track (speakIntoMeeting → recall.outputAudio, ~1.8s
5219
+ // synth, verified audible + clear). The browser CAST is opt-in via
5220
+ // explicit data.castUrl / OSBORN_MEETING_CAST_URL until we split
5221
+ // video(webpage) from audio(output_audio) to reclaim it cleanly.
5222
+ const castUrl = data.castUrl || process.env.OSBORN_MEETING_CAST_URL || undefined;
5219
5223
  const botId = await recallJoin.joinMeeting(meetingUrl, webhookBase, { castUrl });
5220
5224
  const sessionId = currentLLM?.sessionId || currentResumeSessionId || 'default';
5221
5225
  recallJoin.registerBot(botId, sessionId);
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.114",
3
+ "version": "0.9.116",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {