osborn 0.9.112 → 0.9.113

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -230,6 +230,62 @@ let meetingAgentSpeakingText = '';
230
230
  // Prepended to the next flush: what the bot was cut off saying + who interrupted
231
231
  // (same pattern as voice-native interruptions).
232
232
  let meetingInterruptContext = '';
233
+ // Synthesize speech as MP3 (Deepgram fast path, OpenAI fallback) — for
234
+ // Recall native output_audio, which requires mp3.
235
+ async function synthMp3(text) {
236
+ const t0 = Date.now();
237
+ const dgKey = process.env.DEEPGRAM_API_KEY;
238
+ if (dgKey) {
239
+ try {
240
+ const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-thalia-en&encoding=mp3&bit_rate=48000', {
241
+ method: 'POST',
242
+ headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
243
+ body: JSON.stringify({ text: text.slice(0, 4000) }),
244
+ signal: AbortSignal.timeout(12000),
245
+ });
246
+ if (dg.ok) {
247
+ const buf = Buffer.from(await dg.arrayBuffer());
248
+ console.log(`🗣️ synthMp3 deepgram ${buf.length}b in ${Date.now() - t0}ms`);
249
+ return buf;
250
+ }
251
+ }
252
+ catch (e) {
253
+ console.warn(`⚠️ synthMp3 deepgram: ${e.message}`);
254
+ }
255
+ }
256
+ const oa = process.env.OPENAI_API_KEY;
257
+ if (oa) {
258
+ try {
259
+ const r = await fetch('https://api.openai.com/v1/audio/speech', {
260
+ method: 'POST',
261
+ headers: { 'Authorization': `Bearer ${oa}`, 'Content-Type': 'application/json' },
262
+ body: JSON.stringify({ model: 'gpt-4o-mini-tts', voice: 'alloy', input: text.slice(0, 4000), response_format: 'mp3' }),
263
+ signal: AbortSignal.timeout(15000),
264
+ });
265
+ if (r.ok)
266
+ return Buffer.from(await r.arrayBuffer());
267
+ }
268
+ catch { /* fall through */ }
269
+ }
270
+ return null;
271
+ }
272
+ // THE speak path for meetings (2026-08-01): Recall native output_audio FIRST
273
+ // (direct, loud, no canvas capture chain), canvas Web-Audio say as FALLBACK.
274
+ async function speakIntoMeeting(text) {
275
+ const recall = getRecallClient();
276
+ const botId = recall?.getActiveBotIds?.()[0];
277
+ if (recall && botId) {
278
+ const mp3 = await synthMp3(text);
279
+ if (mp3 && await recall.outputAudio(botId, mp3)) {
280
+ console.log(`📢 spoke via Recall output_audio (${mp3.length}b): "${text.slice(0, 60)}"`);
281
+ markMeetingSpeaking(text);
282
+ return;
283
+ }
284
+ }
285
+ console.log(`📽️ falling back to canvas say: "${text.slice(0, 50)}"`);
286
+ pushCanvas({ kind: 'say', text });
287
+ markMeetingSpeaking(text);
288
+ }
233
289
  function markMeetingSpeaking(text) {
234
290
  meetingAgentSpeaking = true;
235
291
  meetingAgentSpeakingText = text;
@@ -644,10 +700,13 @@ function startApiServer(workingDir, port) {
644
700
  const evt = JSON.parse(body || '{}');
645
701
  if (evt.kind !== 'say' && evt.kind !== 'show' && evt.kind !== 'stop')
646
702
  throw new Error("kind must be 'say', 'show', or 'stop'");
647
- pushCanvas(evt);
648
- // Track that the bot is now speaking into the meeting → enables interruption.
649
- if (evt.kind === 'say')
650
- markMeetingSpeaking(evt.text);
703
+ if (evt.kind === 'say') {
704
+ // Native-first speak path (Recall output_audio → canvas fallback).
705
+ void speakIntoMeeting(evt.text);
706
+ }
707
+ else {
708
+ pushCanvas(evt);
709
+ }
651
710
  res.writeHead(200, { 'Content-Type': 'application/json' });
652
711
  res.end(JSON.stringify({ ok: true, clients: canvasClients.size }));
653
712
  }
@@ -1870,6 +1929,16 @@ async function main() {
1870
1929
  // mis-transcriptions; mild false-positive risk is acceptable in a
1871
1930
  // room that invited the bot. AND: in a 1:1 meeting (one human + the
1872
1931
  // bot) every utterance is addressed — no name needed.
1932
+ // Spoken mode switch: "interactive mode"/"go interactive" opens the
1933
+ // conversation latch; "observer mode"/"go silent" closes it.
1934
+ if (/\b(interactive mode|go interactive|start responding|talk to (me|us))\b/i.test(text)) {
1935
+ meetingAddressedUntil = Date.now() + 6 * 60 * 60 * 1000;
1936
+ void speakIntoMeeting('Interactive mode on — I will respond out loud.');
1937
+ }
1938
+ else if (/\b(observer mode|silent mode|go silent|stop responding)\b/i.test(text)) {
1939
+ meetingAddressedUntil = 0;
1940
+ void speakIntoMeeting('Going silent — just taking notes. Say interactive mode to bring me back.');
1941
+ }
1873
1942
  const oneOnOne = meetingSpeakers.size <= 1;
1874
1943
  if (oneOnOne || /\b(osborne?|oz\s?born|os\s?born|was born|is born|ozborn|osbourne?|austin\b.{0,8}(hear|there|can you))/i.test(text)) {
1875
1944
  console.log(`📓 Addressed (${oneOnOne ? '1:1 meeting' : 'by name'}) — immediate flush for a response`);
@@ -2572,9 +2641,8 @@ async function main() {
2572
2641
  // audio file = zero seams — quality over ~1s of latency. (Structural
2573
2642
  // fix queued: canvas as LiveKit subscriber playing the agent's real
2574
2643
  // TTS track — the true "same browser experience" reuse.)
2575
- console.log(`🔊➡️📽️ tts_say → canvas (single utterance) t=${new Date().toISOString()}: "${data.text.slice(0, 60)}"`);
2576
- pushCanvas({ kind: 'say', text: data.text });
2577
- markMeetingSpeaking(data.text);
2644
+ console.log(`🔊➡️📢 tts_say → meeting (native audio) t=${new Date().toISOString()}: "${data.text.slice(0, 60)}"`);
2645
+ void speakIntoMeeting(data.text);
2578
2646
  return;
2579
2647
  }
2580
2648
  console.log(`🔇 tts_say suppressed (meeting turn — response goes to /canvas, not browser): "${data.text.slice(0, 60)}"`);
@@ -103,6 +103,13 @@ export declare class RecallClient extends EventEmitter {
103
103
  */
104
104
  getTranscript(botId: string): Promise<TranscriptTurn[]>;
105
105
  leaveMeeting(botId: string): Promise<void>;
106
+ /**
107
+ * Speak DIRECTLY through the bot via Recall's native output_audio (mp3).
108
+ * No canvas page, no Web Audio, no capture chain — Recall plays the file as
109
+ * the bot's voice. Returns true on 2xx. (2026-08-01: canvas-audio quality
110
+ * was "barely bearable"; this is the direct loud path.)
111
+ */
112
+ outputAudio(botId: string, mp3: Buffer): Promise<boolean>;
106
113
  getBotStatus(botId: string): Promise<string>;
107
114
  handleWebhook(payload: TranscriptPayload): void;
108
115
  registerBot(botId: string, sessionId: string): void;
@@ -146,6 +146,25 @@ export class RecallClient extends EventEmitter {
146
146
  this.#activeBots.delete(botId);
147
147
  console.log(`👋 Recall.ai bot left meeting: ${botId}`);
148
148
  }
149
+ /**
150
+ * Speak DIRECTLY through the bot via Recall's native output_audio (mp3).
151
+ * No canvas page, no Web Audio, no capture chain — Recall plays the file as
152
+ * the bot's voice. Returns true on 2xx. (2026-08-01: canvas-audio quality
153
+ * was "barely bearable"; this is the direct loud path.)
154
+ */
155
+ async outputAudio(botId, mp3) {
156
+ const res = await fetch(`${RECALL_BASE_URL}/bot/${botId}/output_audio/`, {
157
+ method: 'POST',
158
+ headers: { 'Authorization': `Token ${this.#apiKey}`, 'Content-Type': 'application/json' },
159
+ body: JSON.stringify({ kind: 'mp3', b64_data: mp3.toString('base64') }),
160
+ });
161
+ if (!res.ok) {
162
+ const e = await res.text().catch(() => '');
163
+ console.warn(`⚠️ Recall output_audio ${res.status}: ${e.slice(0, 160)}`);
164
+ return false;
165
+ }
166
+ return true;
167
+ }
149
168
  async getBotStatus(botId) {
150
169
  const res = await fetch(`${RECALL_BASE_URL}/bot/${botId}`, {
151
170
  headers: { 'Authorization': `Token ${this.#apiKey}` },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.112",
3
+ "version": "0.9.113",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {