osborn 0.9.111 → 0.9.113

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -230,6 +230,62 @@ let meetingAgentSpeakingText = '';
230
230
  // Prepended to the next flush: what the bot was cut off saying + who interrupted
231
231
  // (same pattern as voice-native interruptions).
232
232
  let meetingInterruptContext = '';
233
+ // Synthesize speech as MP3 (Deepgram fast path, OpenAI fallback) — for
234
+ // Recall native output_audio, which requires mp3.
235
+ async function synthMp3(text) {
236
+ const t0 = Date.now();
237
+ const dgKey = process.env.DEEPGRAM_API_KEY;
238
+ if (dgKey) {
239
+ try {
240
+ const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-thalia-en&encoding=mp3&bit_rate=48000', {
241
+ method: 'POST',
242
+ headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
243
+ body: JSON.stringify({ text: text.slice(0, 4000) }),
244
+ signal: AbortSignal.timeout(12000),
245
+ });
246
+ if (dg.ok) {
247
+ const buf = Buffer.from(await dg.arrayBuffer());
248
+ console.log(`🗣️ synthMp3 deepgram ${buf.length}b in ${Date.now() - t0}ms`);
249
+ return buf;
250
+ }
251
+ }
252
+ catch (e) {
253
+ console.warn(`⚠️ synthMp3 deepgram: ${e.message}`);
254
+ }
255
+ }
256
+ const oa = process.env.OPENAI_API_KEY;
257
+ if (oa) {
258
+ try {
259
+ const r = await fetch('https://api.openai.com/v1/audio/speech', {
260
+ method: 'POST',
261
+ headers: { 'Authorization': `Bearer ${oa}`, 'Content-Type': 'application/json' },
262
+ body: JSON.stringify({ model: 'gpt-4o-mini-tts', voice: 'alloy', input: text.slice(0, 4000), response_format: 'mp3' }),
263
+ signal: AbortSignal.timeout(15000),
264
+ });
265
+ if (r.ok)
266
+ return Buffer.from(await r.arrayBuffer());
267
+ }
268
+ catch { /* fall through */ }
269
+ }
270
+ return null;
271
+ }
272
+ // THE speak path for meetings (2026-08-01): Recall native output_audio FIRST
273
+ // (direct, loud, no canvas capture chain), canvas Web-Audio say as FALLBACK.
274
+ async function speakIntoMeeting(text) {
275
+ const recall = getRecallClient();
276
+ const botId = recall?.getActiveBotIds?.()[0];
277
+ if (recall && botId) {
278
+ const mp3 = await synthMp3(text);
279
+ if (mp3 && await recall.outputAudio(botId, mp3)) {
280
+ console.log(`📢 spoke via Recall output_audio (${mp3.length}b): "${text.slice(0, 60)}"`);
281
+ markMeetingSpeaking(text);
282
+ return;
283
+ }
284
+ }
285
+ console.log(`📽️ falling back to canvas say: "${text.slice(0, 50)}"`);
286
+ pushCanvas({ kind: 'say', text });
287
+ markMeetingSpeaking(text);
288
+ }
233
289
  function markMeetingSpeaking(text) {
234
290
  meetingAgentSpeaking = true;
235
291
  meetingAgentSpeakingText = text;
@@ -644,10 +700,13 @@ function startApiServer(workingDir, port) {
644
700
  const evt = JSON.parse(body || '{}');
645
701
  if (evt.kind !== 'say' && evt.kind !== 'show' && evt.kind !== 'stop')
646
702
  throw new Error("kind must be 'say', 'show', or 'stop'");
647
- pushCanvas(evt);
648
- // Track that the bot is now speaking into the meeting → enables interruption.
649
- if (evt.kind === 'say')
650
- markMeetingSpeaking(evt.text);
703
+ if (evt.kind === 'say') {
704
+ // Native-first speak path (Recall output_audio → canvas fallback).
705
+ void speakIntoMeeting(evt.text);
706
+ }
707
+ else {
708
+ pushCanvas(evt);
709
+ }
651
710
  res.writeHead(200, { 'Content-Type': 'application/json' });
652
711
  res.end(JSON.stringify({ ok: true, clients: canvasClients.size }));
653
712
  }
@@ -1656,6 +1715,11 @@ async function main() {
1656
1715
  // REDIRECTED into the meeting canvas instead of suppressed — the regular
1657
1716
  // voice pipeline reused with the meeting as the sink (latency fix).
1658
1717
  let meetingAddressedUntil = 0;
1718
+ // Distinct human speakers seen in the current meeting. In a 1:1 (one human
1719
+ // + the bot) EVERY utterance is addressed to the bot — no name needed
1720
+ // (2026-08-01: "can you open carfax..." got observer-suppressed because
1721
+ // the fresh session's latch was never opened by the name).
1722
+ const meetingSpeakers = new Set();
1659
1723
  let activeMeetingPoller = null; // Transcript poller bound to that bot
1660
1724
  // LIVE meeting transcript → LLM (buffered webhook finals). See recall.on('transcript').
1661
1725
  const meetingTranscriptBuffer = [];
@@ -1747,6 +1811,7 @@ async function main() {
1747
1811
  }
1748
1812
  activeMeetingBotId = null;
1749
1813
  meetingAddressedUntil = 0;
1814
+ meetingSpeakers.clear();
1750
1815
  // leaveBot=false when Recall itself reported the meeting over (bot already gone).
1751
1816
  if (opts.leaveBot !== false) {
1752
1817
  const recall = getRecallClient();
@@ -1857,13 +1922,26 @@ async function main() {
1857
1922
  // the agent replies out loud into the meeting. This is the chat-mode path:
1858
1923
  // named/asked → prompt response. Un-addressed chunks stay in the silent
1859
1924
  // 20s batch for note-taking. That's the two-mode seam, mechanically.
1925
+ meetingSpeakers.add(speaker);
1860
1926
  // Fuzzy name match: meeting STT routinely mangles "Osborn" — observed
1861
1927
  // live 2026-08-01: "Osborn can you hear me" → "i was born can you hear
1862
1928
  // me" (name trigger missed, bot stayed silent). Accept common
1863
1929
  // mis-transcriptions; mild false-positive risk is acceptable in a
1864
- // room that invited the bot.
1865
- if (/\b(osborne?|oz\s?born|os\s?born|was born|is born|ozborn|osbourne?|austin\b.{0,8}(hear|there|can you))/i.test(text)) {
1866
- console.log('📓 Addressed by name (fuzzy) — immediate flush for a response');
1930
+ // room that invited the bot. AND: in a 1:1 meeting (one human + the
1931
+ // bot) every utterance is addressed — no name needed.
1932
+ // Spoken mode switch: "interactive mode"/"go interactive" opens the
1933
+ // conversation latch; "observer mode"/"go silent" closes it.
1934
+ if (/\b(interactive mode|go interactive|start responding|talk to (me|us))\b/i.test(text)) {
1935
+ meetingAddressedUntil = Date.now() + 6 * 60 * 60 * 1000;
1936
+ void speakIntoMeeting('Interactive mode on — I will respond out loud.');
1937
+ }
1938
+ else if (/\b(observer mode|silent mode|go silent|stop responding)\b/i.test(text)) {
1939
+ meetingAddressedUntil = 0;
1940
+ void speakIntoMeeting('Going silent — just taking notes. Say interactive mode to bring me back.');
1941
+ }
1942
+ const oneOnOne = meetingSpeakers.size <= 1;
1943
+ if (oneOnOne || /\b(osborne?|oz\s?born|os\s?born|was born|is born|ozborn|osbourne?|austin\b.{0,8}(hear|there|can you))/i.test(text)) {
1944
+ console.log(`📓 Addressed (${oneOnOne ? '1:1 meeting' : 'by name'}) — immediate flush for a response`);
1867
1945
  flushMeetingBuffer(botId, true);
1868
1946
  }
1869
1947
  }
@@ -2563,9 +2641,8 @@ async function main() {
2563
2641
  // audio file = zero seams — quality over ~1s of latency. (Structural
2564
2642
  // fix queued: canvas as LiveKit subscriber playing the agent's real
2565
2643
  // TTS track — the true "same browser experience" reuse.)
2566
- console.log(`🔊➡️📽️ tts_say → canvas (single utterance) t=${new Date().toISOString()}: "${data.text.slice(0, 60)}"`);
2567
- pushCanvas({ kind: 'say', text: data.text });
2568
- markMeetingSpeaking(data.text);
2644
+ console.log(`🔊➡️📢 tts_say → meeting (native audio) t=${new Date().toISOString()}: "${data.text.slice(0, 60)}"`);
2645
+ void speakIntoMeeting(data.text);
2569
2646
  return;
2570
2647
  }
2571
2648
  console.log(`🔇 tts_say suppressed (meeting turn — response goes to /canvas, not browser): "${data.text.slice(0, 60)}"`);
@@ -103,6 +103,13 @@ export declare class RecallClient extends EventEmitter {
103
103
  */
104
104
  getTranscript(botId: string): Promise<TranscriptTurn[]>;
105
105
  leaveMeeting(botId: string): Promise<void>;
106
+ /**
107
+ * Speak DIRECTLY through the bot via Recall's native output_audio (mp3).
108
+ * No canvas page, no Web Audio, no capture chain — Recall plays the file as
109
+ * the bot's voice. Returns true on 2xx. (2026-08-01: canvas-audio quality
110
+ * was "barely bearable"; this is the direct loud path.)
111
+ */
112
+ outputAudio(botId: string, mp3: Buffer): Promise<boolean>;
106
113
  getBotStatus(botId: string): Promise<string>;
107
114
  handleWebhook(payload: TranscriptPayload): void;
108
115
  registerBot(botId: string, sessionId: string): void;
@@ -146,6 +146,25 @@ export class RecallClient extends EventEmitter {
146
146
  this.#activeBots.delete(botId);
147
147
  console.log(`👋 Recall.ai bot left meeting: ${botId}`);
148
148
  }
149
+ /**
150
+ * Speak DIRECTLY through the bot via Recall's native output_audio (mp3).
151
+ * No canvas page, no Web Audio, no capture chain — Recall plays the file as
152
+ * the bot's voice. Returns true on 2xx. (2026-08-01: canvas-audio quality
153
+ * was "barely bearable"; this is the direct loud path.)
154
+ */
155
+ async outputAudio(botId, mp3) {
156
+ const res = await fetch(`${RECALL_BASE_URL}/bot/${botId}/output_audio/`, {
157
+ method: 'POST',
158
+ headers: { 'Authorization': `Token ${this.#apiKey}`, 'Content-Type': 'application/json' },
159
+ body: JSON.stringify({ kind: 'mp3', b64_data: mp3.toString('base64') }),
160
+ });
161
+ if (!res.ok) {
162
+ const e = await res.text().catch(() => '');
163
+ console.warn(`⚠️ Recall output_audio ${res.status}: ${e.slice(0, 160)}`);
164
+ return false;
165
+ }
166
+ return true;
167
+ }
149
168
  async getBotStatus(botId) {
150
169
  const res = await fetch(`${RECALL_BASE_URL}/bot/${botId}`, {
151
170
  headers: { 'Authorization': `Token ${this.#apiKey}` },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.111",
3
+ "version": "0.9.113",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {