osborn 0.9.112 → 0.9.114

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -192,6 +192,11 @@ const livekitState = {
192
192
  // connect-with-retry loop) so the module-level HTTP server can drive them.
193
193
  let intentionalLeave = false;
194
194
  let connectRoomHook = null;
195
+ // Mints a LISTEN-ONLY LiveKit token for the meeting canvas so it can join the
196
+ // agent's room and play the agent's REAL TTS audio track — bit-identical to
197
+ // the browser voice experience (2026-08-01 quality architecture; replaces the
198
+ // synth-file chain when connected). Set from main() where the room name lives.
199
+ let canvasTokenHook = null;
195
200
  let leaveRoomHook = null;
196
201
  // Hook for the bug-reporter skill. The /report-bug HTTP endpoint validates the
197
202
  // payload + generates the reportId in the module-level handler, then delegates
@@ -230,6 +235,62 @@ let meetingAgentSpeakingText = '';
230
235
  // Prepended to the next flush: what the bot was cut off saying + who interrupted
231
236
  // (same pattern as voice-native interruptions).
232
237
  let meetingInterruptContext = '';
238
+ // Synthesize speech as MP3 (Deepgram fast path, OpenAI fallback) — for
239
+ // Recall native output_audio, which requires mp3.
240
+ async function synthMp3(text) {
241
+ const t0 = Date.now();
242
+ const dgKey = process.env.DEEPGRAM_API_KEY;
243
+ if (dgKey) {
244
+ try {
245
+ const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-thalia-en&encoding=mp3&bit_rate=48000', {
246
+ method: 'POST',
247
+ headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
248
+ body: JSON.stringify({ text: text.slice(0, 4000) }),
249
+ signal: AbortSignal.timeout(12000),
250
+ });
251
+ if (dg.ok) {
252
+ const buf = Buffer.from(await dg.arrayBuffer());
253
+ console.log(`🗣️ synthMp3 deepgram ${buf.length}b in ${Date.now() - t0}ms`);
254
+ return buf;
255
+ }
256
+ }
257
+ catch (e) {
258
+ console.warn(`⚠️ synthMp3 deepgram: ${e.message}`);
259
+ }
260
+ }
261
+ const oa = process.env.OPENAI_API_KEY;
262
+ if (oa) {
263
+ try {
264
+ const r = await fetch('https://api.openai.com/v1/audio/speech', {
265
+ method: 'POST',
266
+ headers: { 'Authorization': `Bearer ${oa}`, 'Content-Type': 'application/json' },
267
+ body: JSON.stringify({ model: 'gpt-4o-mini-tts', voice: 'alloy', input: text.slice(0, 4000), response_format: 'mp3' }),
268
+ signal: AbortSignal.timeout(15000),
269
+ });
270
+ if (r.ok)
271
+ return Buffer.from(await r.arrayBuffer());
272
+ }
273
+ catch { /* fall through */ }
274
+ }
275
+ return null;
276
+ }
277
+ // THE speak path for meetings (2026-08-01): Recall native output_audio FIRST
278
+ // (direct, loud, no canvas capture chain), canvas Web-Audio say as FALLBACK.
279
+ async function speakIntoMeeting(text) {
280
+ const recall = getRecallClient();
281
+ const botId = recall?.getActiveBotIds?.()[0];
282
+ if (recall && botId) {
283
+ const mp3 = await synthMp3(text);
284
+ if (mp3 && await recall.outputAudio(botId, mp3)) {
285
+ console.log(`📢 spoke via Recall output_audio (${mp3.length}b): "${text.slice(0, 60)}"`);
286
+ markMeetingSpeaking(text);
287
+ return;
288
+ }
289
+ }
290
+ console.log(`📽️ falling back to canvas say: "${text.slice(0, 50)}"`);
291
+ pushCanvas({ kind: 'say', text });
292
+ markMeetingSpeaking(text);
293
+ }
233
294
  function markMeetingSpeaking(text) {
234
295
  meetingAgentSpeaking = true;
235
296
  meetingAgentSpeakingText = text;
@@ -286,6 +347,18 @@ function startApiServer(workingDir, port) {
286
347
  }
287
348
  return;
288
349
  }
350
+ if (req.method === 'GET' && url.pathname === '/canvas-token') {
351
+ // Listen-only LiveKit credentials for the meeting canvas (see hook doc).
352
+ const out = canvasTokenHook ? await canvasTokenHook() : null;
353
+ if (!out) {
354
+ res.writeHead(503, { 'Content-Type': 'application/json' });
355
+ res.end(JSON.stringify({ error: 'no active room' }));
356
+ return;
357
+ }
358
+ res.writeHead(200, { 'Content-Type': 'application/json' });
359
+ res.end(JSON.stringify(out));
360
+ return;
361
+ }
289
362
  if (req.method === 'GET' && url.pathname === '/skills') {
290
363
  // Installed skills — same list the chat's get_skills data-channel message
291
364
  // returns, exposed over HTTP so the DASHBOARD (no LiveKit connection) can
@@ -644,10 +717,13 @@ function startApiServer(workingDir, port) {
644
717
  const evt = JSON.parse(body || '{}');
645
718
  if (evt.kind !== 'say' && evt.kind !== 'show' && evt.kind !== 'stop')
646
719
  throw new Error("kind must be 'say', 'show', or 'stop'");
647
- pushCanvas(evt);
648
- // Track that the bot is now speaking into the meeting → enables interruption.
649
- if (evt.kind === 'say')
650
- markMeetingSpeaking(evt.text);
720
+ if (evt.kind === 'say') {
721
+ // Native-first speak path (Recall output_audio → canvas fallback).
722
+ void speakIntoMeeting(evt.text);
723
+ }
724
+ else {
725
+ pushCanvas(evt);
726
+ }
651
727
  res.writeHead(200, { 'Content-Type': 'application/json' });
652
728
  res.end(JSON.stringify({ ok: true, clients: canvasClients.size }));
653
729
  }
@@ -1656,6 +1732,10 @@ async function main() {
1656
1732
  // REDIRECTED into the meeting canvas instead of suppressed — the regular
1657
1733
  // voice pipeline reused with the meeting as the sink (latency fix).
1658
1734
  let meetingAddressedUntil = 0;
1735
+ // True while the meeting-canvas page is connected to the LiveKit room as a
1736
+ // listener (browser-parity audio path). When set, meeting replies use the
1737
+ // NORMAL session.say pipeline — no suppression, no redirect, no synth chain.
1738
+ let meetingCanvasInRoom = false;
1659
1739
  // Distinct human speakers seen in the current meeting. In a 1:1 (one human
1660
1740
  // + the bot) EVERY utterance is addressed to the bot — no name needed
1661
1741
  // (2026-08-01: "can you open carfax..." got observer-suppressed because
@@ -1870,6 +1950,16 @@ async function main() {
1870
1950
  // mis-transcriptions; mild false-positive risk is acceptable in a
1871
1951
  // room that invited the bot. AND: in a 1:1 meeting (one human + the
1872
1952
  // bot) every utterance is addressed — no name needed.
1953
+ // Spoken mode switch: "interactive mode"/"go interactive" opens the
1954
+ // conversation latch; "observer mode"/"go silent" closes it.
1955
+ if (/\b(interactive mode|go interactive|start responding|talk to (me|us))\b/i.test(text)) {
1956
+ meetingAddressedUntil = Date.now() + 6 * 60 * 60 * 1000;
1957
+ void speakIntoMeeting('Interactive mode on — I will respond out loud.');
1958
+ }
1959
+ else if (/\b(observer mode|silent mode|go silent|stop responding)\b/i.test(text)) {
1960
+ meetingAddressedUntil = 0;
1961
+ void speakIntoMeeting('Going silent — just taking notes. Say interactive mode to bring me back.');
1962
+ }
1873
1963
  const oneOnOne = meetingSpeakers.size <= 1;
1874
1964
  if (oneOnOne || /\b(osborne?|oz\s?born|os\s?born|was born|is born|ozborn|osbourne?|austin\b.{0,8}(hear|there|can you))/i.test(text)) {
1875
1965
  console.log(`📓 Addressed (${oneOnOne ? '1:1 meeting' : 'by name'}) — immediate flush for a response`);
@@ -2563,6 +2653,16 @@ async function main() {
2563
2653
  // re-captures in the same room → feedback). Set by PipelineDirectLLM.chat()
2564
2654
  // when the turn is a [MEETING —] chunk. Normal user turns are unaffected.
2565
2655
  if (directLLM.suppressMeetingTTS) {
2656
+ // BROWSER-PARITY PATH: canvas is in the LiveKit room → the normal
2657
+ // session.say audio reaches the meeting through it. Fall through to
2658
+ // the regular pipeline for addressed turns (identical audio to the
2659
+ // browser experience); observer turns still stay silent.
2660
+ if (meetingCanvasInRoom && activeMeetingBotId && Date.now() < meetingAddressedUntil) {
2661
+ console.log(`🔊🎼 meeting reply via NATIVE session.say (canvas relays): "${data.text.slice(0, 50)}"`);
2662
+ markMeetingSpeaking(data.text);
2663
+ // no return — normal say proceeds below
2664
+ }
2665
+ else
2566
2666
  // ADDRESSED turn → REDIRECT the normal streaming reply into the meeting
2567
2667
  // (reuse of the regular speak path; no Bash roundtrip). Silent-observer
2568
2668
  // turns stay suppressed as before.
@@ -2572,9 +2672,8 @@ async function main() {
2572
2672
  // audio file = zero seams — quality over ~1s of latency. (Structural
2573
2673
  // fix queued: canvas as LiveKit subscriber playing the agent's real
2574
2674
  // TTS track — the true "same browser experience" reuse.)
2575
- console.log(`🔊➡️📽️ tts_say → canvas (single utterance) t=${new Date().toISOString()}: "${data.text.slice(0, 60)}"`);
2576
- pushCanvas({ kind: 'say', text: data.text });
2577
- markMeetingSpeaking(data.text);
2675
+ console.log(`🔊➡️📢 tts_say → meeting (native audio) t=${new Date().toISOString()}: "${data.text.slice(0, 60)}"`);
2676
+ void speakIntoMeeting(data.text);
2578
2677
  return;
2579
2678
  }
2580
2679
  console.log(`🔇 tts_say suppressed (meeting turn — response goes to /canvas, not browser): "${data.text.slice(0, 60)}"`);
@@ -3510,6 +3609,14 @@ async function main() {
3510
3609
  // connect callback, not via events).
3511
3610
  participantConnectedHandler = async (participant) => {
3512
3611
  console.log(`\n👤 User joined: ${participant.identity}`);
3612
+ // The meeting canvas joining as a listener = browser-parity audio path is
3613
+ // LIVE: the agent's normal session.say plays through the canvas into the
3614
+ // meeting, so tts_say suppression/redirect must stand down.
3615
+ if (participant.identity === 'meeting-canvas') {
3616
+ meetingCanvasInRoom = true;
3617
+ console.log('📽️🔊 meeting-canvas joined the LiveKit room — native session.say relays to the meeting');
3618
+ return; // not a user; skip session setup for it
3619
+ }
3513
3620
  // A user (re)arriving cancels the meeting leave-grace — the bot stays.
3514
3621
  cancelMeetingLeaveGrace();
3515
3622
  // A user is present — cancel any pending agent-side "alone" leave.
@@ -5413,6 +5520,17 @@ async function main() {
5413
5520
  // { roomName } (which connectRoomHook resolves to). During rollout, the legacy
5414
5521
  // GET /room-code endpoint keeps returning the last-created room NAME so old
5415
5522
  // frontends still function.
5523
+ canvasTokenHook = async () => {
5524
+ if (!activeRoomName)
5525
+ return null;
5526
+ const t = new AccessToken(apiKey, apiSecret, {
5527
+ identity: 'meeting-canvas',
5528
+ name: 'Meeting Canvas',
5529
+ metadata: JSON.stringify({ type: 'canvas' }),
5530
+ });
5531
+ t.addGrant({ roomJoin: true, room: activeRoomName, canPublish: false, canSubscribe: true, canPublishData: false });
5532
+ return { token: await t.toJwt(), url: process.env.LIVEKIT_URL || '', room: activeRoomName };
5533
+ };
5416
5534
  connectRoomHook = async () => {
5417
5535
  intentionalLeave = false;
5418
5536
  cancelIdleExitTimer();
@@ -103,6 +103,13 @@ export declare class RecallClient extends EventEmitter {
103
103
  */
104
104
  getTranscript(botId: string): Promise<TranscriptTurn[]>;
105
105
  leaveMeeting(botId: string): Promise<void>;
106
+ /**
107
+ * Speak DIRECTLY through the bot via Recall's native output_audio (mp3).
108
+ * No canvas page, no Web Audio, no capture chain — Recall plays the file as
109
+ * the bot's voice. Returns true on 2xx. (2026-08-01: canvas-audio quality
110
+ * was "barely bearable"; this is the direct loud path.)
111
+ */
112
+ outputAudio(botId: string, mp3: Buffer): Promise<boolean>;
106
113
  getBotStatus(botId: string): Promise<string>;
107
114
  handleWebhook(payload: TranscriptPayload): void;
108
115
  registerBot(botId: string, sessionId: string): void;
@@ -146,6 +146,25 @@ export class RecallClient extends EventEmitter {
146
146
  this.#activeBots.delete(botId);
147
147
  console.log(`👋 Recall.ai bot left meeting: ${botId}`);
148
148
  }
149
+ /**
150
+ * Speak DIRECTLY through the bot via Recall's native output_audio (mp3).
151
+ * No canvas page, no Web Audio, no capture chain — Recall plays the file as
152
+ * the bot's voice. Returns true on 2xx. (2026-08-01: canvas-audio quality
153
+ * was "barely bearable"; this is the direct loud path.)
154
+ */
155
+ async outputAudio(botId, mp3) {
156
+ const res = await fetch(`${RECALL_BASE_URL}/bot/${botId}/output_audio/`, {
157
+ method: 'POST',
158
+ headers: { 'Authorization': `Token ${this.#apiKey}`, 'Content-Type': 'application/json' },
159
+ body: JSON.stringify({ kind: 'mp3', b64_data: mp3.toString('base64') }),
160
+ });
161
+ if (!res.ok) {
162
+ const e = await res.text().catch(() => '');
163
+ console.warn(`⚠️ Recall output_audio ${res.status}: ${e.slice(0, 160)}`);
164
+ return false;
165
+ }
166
+ return true;
167
+ }
149
168
  async getBotStatus(botId) {
150
169
  const res = await fetch(`${RECALL_BASE_URL}/bot/${botId}`, {
151
170
  headers: { 'Authorization': `Token ${this.#apiKey}` },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.112",
3
+ "version": "0.9.114",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {