osborn 0.9.115 → 0.9.117
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +17 -9
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -276,6 +276,12 @@ async function synthMp3(text) {
|
|
|
276
276
|
}
|
|
277
277
|
// THE speak path for meetings (2026-08-01): Recall native output_audio FIRST
|
|
278
278
|
// (direct, loud, no canvas capture chain), canvas Web-Audio say as FALLBACK.
|
|
279
|
+
// THE meeting speak path (2026-08-01, live-verified A/B): VOICE via Recall
|
|
280
|
+
// output_audio (reliable, doesn't depend on the headless canvas AudioContext),
|
|
281
|
+
// VISUAL caption pushed to the canvas WITHOUT audio (kind:'caption') so the
|
|
282
|
+
// bot's camera shows what it's saying — no double-audio. output_audio +
|
|
283
|
+
// canvas camera coexist (confirmed: user heard output_audio while the canvas
|
|
284
|
+
// was showing). Falls back to canvas 'say' (audio) only if output_audio fails.
|
|
279
285
|
async function speakIntoMeeting(text) {
|
|
280
286
|
const recall = getRecallClient();
|
|
281
287
|
const botId = recall?.getActiveBotIds?.()[0];
|
|
@@ -283,11 +289,12 @@ async function speakIntoMeeting(text) {
|
|
|
283
289
|
const mp3 = await synthMp3(text);
|
|
284
290
|
if (mp3 && await recall.outputAudio(botId, mp3)) {
|
|
285
291
|
console.log(`📢 spoke via Recall output_audio (${mp3.length}b): "${text.slice(0, 60)}"`);
|
|
292
|
+
pushCanvas({ kind: 'caption', text }); // visual only, no audio
|
|
286
293
|
markMeetingSpeaking(text);
|
|
287
294
|
return;
|
|
288
295
|
}
|
|
289
296
|
}
|
|
290
|
-
console.log(`📽️ falling back to canvas say: "${text.slice(0, 50)}"`);
|
|
297
|
+
console.log(`📽️ falling back to canvas say (audio): "${text.slice(0, 50)}"`);
|
|
291
298
|
pushCanvas({ kind: 'say', text });
|
|
292
299
|
markMeetingSpeaking(text);
|
|
293
300
|
}
|
|
@@ -715,8 +722,8 @@ function startApiServer(workingDir, port) {
|
|
|
715
722
|
req.on('end', () => {
|
|
716
723
|
try {
|
|
717
724
|
const evt = JSON.parse(body || '{}');
|
|
718
|
-
if (evt.kind !== 'say' && evt.kind !== 'show' && evt.kind !== 'stop')
|
|
719
|
-
throw new Error("kind must be 'say', 'show', or 'stop'");
|
|
725
|
+
if (evt.kind !== 'say' && evt.kind !== 'caption' && evt.kind !== 'show' && evt.kind !== 'stop')
|
|
726
|
+
throw new Error("kind must be 'say', 'caption', 'show', or 'stop'");
|
|
720
727
|
if (evt.kind === 'say') {
|
|
721
728
|
// Native-first speak path (Recall output_audio → canvas fallback).
|
|
722
729
|
void speakIntoMeeting(evt.text);
|
|
@@ -5210,12 +5217,13 @@ async function main() {
|
|
|
5210
5217
|
// the bot casts the meeting canvas as its camera+mic (below), so
|
|
5211
5218
|
// it can show visuals and speak into the meeting on demand.
|
|
5212
5219
|
await sendToFrontend({ type: 'meeting_joining', message: 'Osborn is joining your meeting...' });
|
|
5213
|
-
//
|
|
5214
|
-
//
|
|
5215
|
-
//
|
|
5216
|
-
//
|
|
5217
|
-
//
|
|
5218
|
-
//
|
|
5220
|
+
// CANVAS + AUDIO (2026-08-01, live-verified A/B): the canvas
|
|
5221
|
+
// webpage camera shows VISUALS (captions, screenshots, transcript
|
|
5222
|
+
// animation) AND Recall output_audio delivers the VOICE — the two
|
|
5223
|
+
// coexist (confirmed: output_audio audible while the canvas camera
|
|
5224
|
+
// was showing). speakIntoMeeting routes voice→output_audio +
|
|
5225
|
+
// caption→canvas (no audio), so no double-audio. The canvas page's
|
|
5226
|
+
// OWN Web Audio is NOT used for voice (it's suspended headless).
|
|
5219
5227
|
const frontendUrl = (process.env.OSBORN_FRONTEND_URL || 'https://www.voice-native.com').replace(/\/$/, '');
|
|
5220
5228
|
const canvasUrl = /^https:\/\//.test(webhookBase)
|
|
5221
5229
|
? `${frontendUrl}/meeting-canvas?agent=${encodeURIComponent(webhookBase)}`
|