osborn 0.9.116 → 0.9.117

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/dist/index.js +22 -13
  2. package/package.json +1 -1
package/dist/index.js CHANGED
@@ -276,6 +276,12 @@ async function synthMp3(text) {
276
276
  }
277
277
  // THE speak path for meetings (2026-08-01): Recall native output_audio FIRST
278
278
  // (direct, loud, no canvas capture chain), canvas Web-Audio say as FALLBACK.
279
+ // THE meeting speak path (2026-08-01, live-verified A/B): VOICE via Recall
280
+ // output_audio (reliable, doesn't depend on the headless canvas AudioContext),
281
+ // VISUAL caption pushed to the canvas WITHOUT audio (kind:'caption') so the
282
+ // bot's camera shows what it's saying — no double-audio. output_audio +
283
+ // canvas camera coexist (confirmed: user heard output_audio while the canvas
284
+ // was showing). Falls back to canvas 'say' (audio) only if output_audio fails.
279
285
  async function speakIntoMeeting(text) {
280
286
  const recall = getRecallClient();
281
287
  const botId = recall?.getActiveBotIds?.()[0];
@@ -283,11 +289,12 @@ async function speakIntoMeeting(text) {
283
289
  const mp3 = await synthMp3(text);
284
290
  if (mp3 && await recall.outputAudio(botId, mp3)) {
285
291
  console.log(`📢 spoke via Recall output_audio (${mp3.length}b): "${text.slice(0, 60)}"`);
292
+ pushCanvas({ kind: 'caption', text }); // visual only, no audio
286
293
  markMeetingSpeaking(text);
287
294
  return;
288
295
  }
289
296
  }
290
- console.log(`📽️ falling back to canvas say: "${text.slice(0, 50)}"`);
297
+ console.log(`📽️ falling back to canvas say (audio): "${text.slice(0, 50)}"`);
291
298
  pushCanvas({ kind: 'say', text });
292
299
  markMeetingSpeaking(text);
293
300
  }
@@ -715,8 +722,8 @@ function startApiServer(workingDir, port) {
715
722
  req.on('end', () => {
716
723
  try {
717
724
  const evt = JSON.parse(body || '{}');
718
- if (evt.kind !== 'say' && evt.kind !== 'show' && evt.kind !== 'stop')
719
- throw new Error("kind must be 'say', 'show', or 'stop'");
725
+ if (evt.kind !== 'say' && evt.kind !== 'caption' && evt.kind !== 'show' && evt.kind !== 'stop')
726
+ throw new Error("kind must be 'say', 'caption', 'show', or 'stop'");
720
727
  if (evt.kind === 'say') {
721
728
  // Native-first speak path (Recall output_audio → canvas fallback).
722
729
  void speakIntoMeeting(evt.text);
@@ -5210,16 +5217,18 @@ async function main() {
5210
5217
  // the bot casts the meeting canvas as its camera+mic (below), so
5211
5218
  // it can show visuals and speak into the meeting on demand.
5212
5219
  await sendToFrontend({ type: 'meeting_joining', message: 'Osborn is joining your meeting...' });
5213
- // AUDIO-FIRST DEFAULT (2026-08-01, proven live): the webpage-camera
5214
- // output_media routes the bot's AUDIO through the canvas page's
5215
- // Web Audio — which sits SUSPENDED in Recall's headless Chrome (no
5216
- // user gesture) → the bot was inaudible all night. Default now = NO
5217
- // webpage camera, so the bot's voice is the direct Recall
5218
- // output_audio track (speakIntoMeeting → recall.outputAudio, ~1.8s
5219
- // synth, verified audible + clear). The browser CAST is opt-in via
5220
- // explicit data.castUrl / OSBORN_MEETING_CAST_URL until we split
5221
- // video(webpage) from audio(output_audio) to reclaim it cleanly.
5222
- const castUrl = data.castUrl || process.env.OSBORN_MEETING_CAST_URL || undefined;
5220
+ // CANVAS + AUDIO (2026-08-01, live-verified A/B): the canvas
5221
+ // webpage camera shows VISUALS (captions, screenshots, transcript
5222
+ // animation) AND Recall output_audio delivers the VOICE — the two
5223
+ // coexist (confirmed: output_audio audible while the canvas camera
5224
+ // was showing). speakIntoMeeting routes voice→output_audio +
5225
+ // caption→canvas (no audio), so no double-audio. The canvas page's
5226
+ // OWN Web Audio is NOT used for voice (it's suspended headless).
5227
+ const frontendUrl = (process.env.OSBORN_FRONTEND_URL || 'https://www.voice-native.com').replace(/\/$/, '');
5228
+ const canvasUrl = /^https:\/\//.test(webhookBase)
5229
+ ? `${frontendUrl}/meeting-canvas?agent=${encodeURIComponent(webhookBase)}`
5230
+ : undefined;
5231
+ const castUrl = data.castUrl || process.env.OSBORN_MEETING_CAST_URL || canvasUrl;
5223
5232
  const botId = await recallJoin.joinMeeting(meetingUrl, webhookBase, { castUrl });
5224
5233
  const sessionId = currentLLM?.sessionId || currentResumeSessionId || 'default';
5225
5234
  recallJoin.registerBot(botId, sessionId);
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.116",
3
+ "version": "0.9.117",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {