osborn 0.9.120 → 0.9.122

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -235,42 +235,71 @@ let meetingAgentSpeakingText = '';
235
235
  // Prepended to the next flush: what the bot was cut off saying + who interrupted
236
236
  // (same pattern as voice-native interruptions).
237
237
  let meetingInterruptContext = '';
238
+ // Meeting speech QUEUE (0.9.121): serialize output_audio so replies never
239
+ // overlap — the Recall-sink equivalent of session.say's SpeechHandle queue.
240
+ // Recall's output_audio POST returns on ACCEPT, not on finish, so without this
241
+ // two replies (from two flushes, or a streamed multi-chunk reply) play ON TOP
242
+ // of each other — the "another voice over it" the user heard. Each utterance
243
+ // waits for the prior one's estimated playback before it plays; a generation
244
+ // counter (bumped on human interruption / a superseding turn) discards anything
245
+ // still queued so the bot never talks over itself or a human.
246
+ let meetingSpeakChain = Promise.resolve();
247
+ let meetingSpeakGen = 0;
248
+ const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
249
+ // Estimated playback duration of a spoken line: ~2.5 words/sec + ~0.8s Recall
250
+ // buffer. Used to hold the speech queue so the next utterance doesn't overlap.
251
+ function estimatedSpeechMs(text) {
252
+ const words = text.split(/\s+/).filter(Boolean).length;
253
+ return Math.min(30_000, 800 + (words / 2.5) * 1000);
254
+ }
255
+ // Interrupt all meeting speech: bump the generation (drops queued + in-synth
256
+ // utterances) and stop any output_audio Recall is currently playing.
257
+ function interruptMeetingSpeech(reason) {
258
+ meetingSpeakGen++;
259
+ meetingAgentSpeaking = false;
260
+ if (meetingSpeakClearTimer) {
261
+ clearTimeout(meetingSpeakClearTimer);
262
+ meetingSpeakClearTimer = null;
263
+ }
264
+ const recall = getRecallClient();
265
+ const botId = recall?.getActiveBotIds?.()[0];
266
+ if (recall && botId)
267
+ void recall.stopOutputAudio(botId);
268
+ console.log(`✋ meeting speech interrupted (${reason}) — queue cleared + output_audio stopped`);
269
+ }
238
270
  // Synthesize speech as MP3 (Deepgram fast path, OpenAI fallback) — for
239
271
  // Recall native output_audio, which requires mp3.
240
272
  async function synthMp3(text) {
241
273
  const t0 = Date.now();
242
- const dgKey = process.env.DEEPGRAM_API_KEY;
243
- if (dgKey) {
244
- try {
245
- const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-asteria-en&encoding=mp3&bit_rate=48000', {
246
- method: 'POST',
247
- headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
248
- body: JSON.stringify({ text: text.slice(0, 4000) }),
249
- signal: AbortSignal.timeout(12000),
250
- });
251
- if (dg.ok) {
252
- const buf = Buffer.from(await dg.arrayBuffer());
253
- console.log(`🗣️ synthMp3 deepgram ${buf.length}b in ${Date.now() - t0}ms`);
254
- return buf;
255
- }
256
- }
257
- catch (e) {
258
- console.warn(`⚠️ synthMp3 deepgram: ${e.message}`);
259
- }
260
- }
261
274
  const oa = process.env.OPENAI_API_KEY;
262
- if (oa) {
263
- try {
264
- const r = await fetch('https://api.openai.com/v1/audio/speech', {
265
- method: 'POST',
266
- headers: { 'Authorization': `Bearer ${oa}`, 'Content-Type': 'application/json' },
267
- body: JSON.stringify({ model: 'gpt-4o-mini-tts', voice: 'alloy', input: text.slice(0, 4000), response_format: 'mp3' }),
268
- signal: AbortSignal.timeout(15000),
269
- });
270
- if (r.ok)
271
- return Buffer.from(await r.arrayBuffer());
275
+ if (!oa) {
276
+ console.warn('⚠️ synthMp3: no OPENAI_API_KEY — meeting has no voice');
277
+ return null;
278
+ }
279
+ // Meeting voice = the SAME OpenAI model/voice as the website's regular TTS
280
+ // (DIRECT_MODE_TTS), so the bot sounds IDENTICAL on both fronts (user directive
281
+ // 2026-08-04: Deepgram aura sounded "cheap and inconsistent"). Deepgram removed
282
+ // from the meeting path entirely. Pulls model/voice from DIRECT_MODE_TTS when
283
+ // it's an OpenAI config so the two never drift.
284
+ const model = DIRECT_MODE_TTS.provider === 'openai' ? (DIRECT_MODE_TTS.model || 'tts-1-hd') : 'tts-1-hd';
285
+ const voice = DIRECT_MODE_TTS.provider === 'openai' ? (DIRECT_MODE_TTS.voice || 'fable') : 'fable';
286
+ try {
287
+ const r = await fetch('https://api.openai.com/v1/audio/speech', {
288
+ method: 'POST',
289
+ headers: { 'Authorization': `Bearer ${oa}`, 'Content-Type': 'application/json' },
290
+ body: JSON.stringify({ model, voice, input: text.slice(0, 4000), response_format: 'mp3' }),
291
+ signal: AbortSignal.timeout(20000),
292
+ });
293
+ if (r.ok) {
294
+ const buf = Buffer.from(await r.arrayBuffer());
295
+ console.log(`🗣️ synthMp3 openai ${model}/${voice} ${buf.length}b in ${Date.now() - t0}ms`);
296
+ return buf;
272
297
  }
273
- catch { /* fall through */ }
298
+ const e = await r.text().catch(() => '');
299
+ console.warn(`⚠️ synthMp3 openai ${r.status}: ${e.slice(0, 120)}`);
300
+ }
301
+ catch (e) {
302
+ console.warn(`⚠️ synthMp3 openai: ${e.message}`);
274
303
  }
275
304
  return null;
276
305
  }
@@ -282,21 +311,43 @@ async function synthMp3(text) {
282
311
  // bot's camera shows what it's saying — no double-audio. output_audio +
283
312
  // canvas camera coexist (confirmed: user heard output_audio while the canvas
284
313
  // was showing). Falls back to canvas 'say' (audio) only if output_audio fails.
285
- async function speakIntoMeeting(text) {
286
- const recall = getRecallClient();
287
- const botId = recall?.getActiveBotIds?.()[0];
288
- if (recall && botId) {
289
- const mp3 = await synthMp3(text);
290
- if (mp3 && await recall.outputAudio(botId, mp3)) {
291
- console.log(`📢 spoke via Recall output_audio (${mp3.length}b): "${text.slice(0, 60)}"`);
292
- pushCanvas({ kind: 'caption', text }); // visual only, no audio
293
- markMeetingSpeaking(text);
314
+ function speakIntoMeeting(text) {
315
+ if (!text?.trim())
316
+ return Promise.resolve();
317
+ // Capture the generation at ENQUEUE time. If an interrupt (or a superseding
318
+ // turn) bumps the gen before this item runs — or mid-synth — we drop it, so
319
+ // the bot never plays a reply the conversation has already moved past.
320
+ const gen = meetingSpeakGen;
321
+ const run = meetingSpeakChain.then(async () => {
322
+ if (gen !== meetingSpeakGen) {
323
+ console.log(`🔇 meeting speech superseded — dropping: "${text.slice(0, 40)}"`);
294
324
  return;
295
325
  }
296
- }
297
- console.log(`📽️ falling back to canvas say (audio): "${text.slice(0, 50)}"`);
298
- pushCanvas({ kind: 'say', text });
299
- markMeetingSpeaking(text);
326
+ const recall = getRecallClient();
327
+ const botId = recall?.getActiveBotIds?.()[0];
328
+ if (recall && botId) {
329
+ const mp3 = await synthMp3(text);
330
+ if (gen !== meetingSpeakGen) {
331
+ console.log(`🔇 meeting speech interrupted mid-synth — dropping: "${text.slice(0, 40)}"`);
332
+ return;
333
+ }
334
+ if (mp3 && await recall.outputAudio(botId, mp3)) {
335
+ console.log(`📢 spoke via Recall output_audio (${mp3.length}b): "${text.slice(0, 60)}"`);
336
+ pushCanvas({ kind: 'caption', text }); // visual only, no audio
337
+ markMeetingSpeaking(text);
338
+ // Hold the queue for the estimated playback so the NEXT utterance
339
+ // doesn't start on top of this one (POST returns on accept, not finish).
340
+ await sleep(estimatedSpeechMs(text));
341
+ return;
342
+ }
343
+ }
344
+ console.log(`📽️ falling back to canvas say (audio): "${text.slice(0, 50)}"`);
345
+ pushCanvas({ kind: 'say', text });
346
+ markMeetingSpeaking(text);
347
+ await sleep(estimatedSpeechMs(text));
348
+ }).catch((e) => { console.warn(`⚠️ meeting speak failed: ${e.message}`); });
349
+ meetingSpeakChain = run;
350
+ return run;
300
351
  }
301
352
  function markMeetingSpeaking(text) {
302
353
  meetingAgentSpeaking = true;
@@ -656,46 +707,23 @@ function startApiServer(workingDir, port) {
656
707
  return;
657
708
  }
658
709
  const t0 = Date.now();
659
- // Deepgram FIRST (2026-08-01 latency fix): same TTS family the regular
660
- // voice pipeline uses — ~3-6x faster than the OpenAI full-file synth this
661
- // endpoint used before (measured 2-4s of the meeting reply lag). OpenAI
662
- // stays as fallback.
663
- const dgKey = process.env.DEEPGRAM_API_KEY;
664
- if (dgKey) {
665
- try {
666
- // WAV/linear16, not mp3: mp3 files carry encoder padding (leading/
667
- // trailing silence + boundary click) — back-to-back sentence clips
668
- // were heard as "cracking". WAV is gapless-safe and cheaper to decode.
669
- const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-asteria-en&encoding=linear16&sample_rate=48000&container=wav', {
670
- method: 'POST',
671
- headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
672
- body: JSON.stringify({ text }),
673
- });
674
- if (dg.ok) {
675
- const buf = Buffer.from(await dg.arrayBuffer());
676
- console.log(`🗣️ /tts deepgram wav ${buf.length}b in ${Date.now() - t0}ms (${text.length} chars) t=${new Date().toISOString()}`);
677
- res.writeHead(200, { 'Content-Type': 'audio/wav', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
678
- res.end(buf);
679
- return;
680
- }
681
- console.warn(`⚠️ /tts deepgram ${dg.status} — falling back to OpenAI`);
682
- }
683
- catch (e) {
684
- console.warn(`⚠️ /tts deepgram error: ${e.message} — falling back to OpenAI`);
685
- }
686
- }
687
- const voice = url.searchParams.get('voice') || 'alloy';
710
+ // Meeting voice = the SAME OpenAI model/voice as the website's regular TTS
711
+ // (DIRECT_MODE_TTS) — user directive 2026-08-04: Deepgram aura removed, it
712
+ // sounded cheap/inconsistent. Consistency over the ~2-4s latency Deepgram
713
+ // saved. mp3 out (the canvas <audio> element plays it into the meeting).
688
714
  const key = process.env.OPENAI_API_KEY;
689
715
  if (!key) {
690
716
  res.writeHead(400, { 'Content-Type': 'application/json' });
691
- res.end(JSON.stringify({ error: 'no TTS provider keys' }));
717
+ res.end(JSON.stringify({ error: 'no OPENAI_API_KEY' }));
692
718
  return;
693
719
  }
720
+ const model = DIRECT_MODE_TTS.provider === 'openai' ? (DIRECT_MODE_TTS.model || 'tts-1-hd') : 'tts-1-hd';
721
+ const voice = url.searchParams.get('voice') || (DIRECT_MODE_TTS.provider === 'openai' ? (DIRECT_MODE_TTS.voice || 'fable') : 'fable');
694
722
  try {
695
723
  const tts = await fetch('https://api.openai.com/v1/audio/speech', {
696
724
  method: 'POST',
697
725
  headers: { 'Authorization': `Bearer ${key}`, 'Content-Type': 'application/json' },
698
- body: JSON.stringify({ model: 'gpt-4o-mini-tts', voice, input: text, response_format: 'mp3' }),
726
+ body: JSON.stringify({ model, voice, input: text, response_format: 'mp3' }),
699
727
  });
700
728
  if (!tts.ok) {
701
729
  const e = await tts.text().catch(() => '');
@@ -704,7 +732,7 @@ function startApiServer(workingDir, port) {
704
732
  return;
705
733
  }
706
734
  const buf = Buffer.from(await tts.arrayBuffer());
707
- console.log(`🗣️ /tts openai ${buf.length}b in ${Date.now() - t0}ms t=${new Date().toISOString()}`);
735
+ console.log(`🗣️ /tts openai ${model}/${voice} ${buf.length}b in ${Date.now() - t0}ms t=${new Date().toISOString()}`);
708
736
  res.writeHead(200, { 'Content-Type': 'audio/mpeg', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
709
737
  res.end(buf);
710
738
  }
@@ -1775,13 +1803,14 @@ async function main() {
1775
1803
  // from muting the audio path. Reset by endMeeting.
1776
1804
  meetingAddressedUntil = Date.now() + 6 * 60 * 60 * 1000;
1777
1805
  }
1778
- // PROMPT PARITY (user directive 2026-08-01): addressed turns carry ONLY a
1779
- // minimal tag — no behavioral re-instruction. The agent replies exactly as
1780
- // it would to a regular voice turn; the tts_say→canvas redirect handles
1781
- // where the words go. Same agent, same behavior, both fronts.
1782
- const header = addressed
1783
- ? `[MEETING — ${botId}] (addressed — reply is spoken into the meeting):`
1784
- : `[MEETING — ${botId}]:`;
1806
+ // NO meeting-specific reply coaching (user directive 2026-08-04): the reply
1807
+ // must be exactly what the main Claude Code agent would say — no brevity
1808
+ // rules, no "spoken out loud" framing. Just a bare [MEETING — id] routing
1809
+ // tag, which is all the plumbing needs: PipelineDirectLLM keys
1810
+ // suppressMeetingTTS off the `[MEETING` prefix, and whether the reply is
1811
+ // SPOKEN is decided in CODE (activeMeetingBotId + meetingAddressedUntil),
1812
+ // not by any prompt instruction. Addressed vs observer is the code boolean.
1813
+ const header = `[MEETING — ${botId}]:`;
1785
1814
  // Prepend + consume any interruption context (bot was cut off mid-sentence).
1786
1815
  const interrupt = meetingInterruptContext ? `${meetingInterruptContext}\n` : '';
1787
1816
  meetingInterruptContext = '';
@@ -1806,8 +1835,33 @@ async function main() {
1806
1835
  meetingFlushTimer = null;
1807
1836
  console.log('📓 Meeting flush timer stopped');
1808
1837
  }
1838
+ if (addressedFlushTimer) {
1839
+ clearTimeout(addressedFlushTimer);
1840
+ addressedFlushTimer = null;
1841
+ }
1809
1842
  meetingTranscriptBuffer.length = 0;
1810
1843
  };
1844
+ // TURN-DEBOUNCED addressed flush (0.9.121): Recall closes a transcript
1845
+ // segment on every pause, so flushing on each final made the bot reply to
1846
+ // FRAGMENTS mid-thought (and multiple times per utterance). Instead we
1847
+ // debounce: each new transcript final resets a short timer; we only flush
1848
+ // (= reply) after the speaker has actually paused, so one turn = one reply.
1849
+ // speech_off (a hard silence boundary) can flush sooner via a smaller delay.
1850
+ let addressedFlushTimer = null;
1851
+ const ADDRESSED_DEBOUNCE_MS = 1400;
1852
+ const scheduleAddressedFlush = (botId, delayMs = ADDRESSED_DEBOUNCE_MS) => {
1853
+ if (addressedFlushTimer)
1854
+ clearTimeout(addressedFlushTimer);
1855
+ addressedFlushTimer = setTimeout(() => {
1856
+ addressedFlushTimer = null;
1857
+ if (!meetingTranscriptBuffer.length)
1858
+ return;
1859
+ // A fresh human turn supersedes anything the bot was still saying about
1860
+ // the previous one — interrupt stale queued/playing speech before we reply.
1861
+ interruptMeetingSpeech('new addressed turn');
1862
+ flushMeetingBuffer(botId, true);
1863
+ }, delayMs);
1864
+ };
1811
1865
  // ── Meeting lifecycle (centralized teardown, 0.9.95) ──
1812
1866
  // The bot's lifecycle FOLLOWS the voice session (deliberate coupling — a
1813
1867
  // decoupled always-on bot means untracked background agents; revisit only
@@ -1969,8 +2023,12 @@ async function main() {
1969
2023
  }
1970
2024
  const oneOnOne = meetingSpeakers.size <= 1;
1971
2025
  if (oneOnOne || /\b(osborne?|oz\s?born|os\s?born|was born|is born|ozborn|osbourne?|austin\b.{0,8}(hear|there|can you))/i.test(text)) {
1972
- console.log(`📓 Addressed (${oneOnOne ? '1:1 meeting' : 'by name'}) — immediate flush for a response`);
1973
- flushMeetingBuffer(botId, true);
2026
+ // DEBOUNCED (0.9.121): don't reply to this fragment — wait for the
2027
+ // speaker to actually pause. Each new final resets the timer, so one
2028
+ // continuous thought (even across Recall's mid-sentence segment splits)
2029
+ // becomes ONE reply instead of several talking over each other.
2030
+ console.log(`📓 Addressed (${oneOnOne ? '1:1 meeting' : 'by name'}) — turn debounced (~${ADDRESSED_DEBOUNCE_MS}ms)`);
2031
+ scheduleAddressedFlush(botId);
1974
2032
  }
1975
2033
  }
1976
2034
  });
@@ -1982,12 +2040,16 @@ async function main() {
1982
2040
  // bot's audio and record what it was cut off saying so the next flush can
1983
2041
  // tell it what it missed (same pattern as voice-native interruptions).
1984
2042
  if (meetingAgentSpeaking && !isBot) {
1985
- console.log(`✋ Interruption — ${participant} spoke while bot was talking. Stopping bot audio.`);
1986
- pushCanvas({ kind: 'stop' });
2043
+ console.log(`✋ Interruption — ${participant} spoke while bot was talking.`);
2044
+ // Capture what got cut off BEFORE interrupting (same interruption-
2045
+ // context ledger as the website path — the next flush tells the agent
2046
+ // it was cut off + what the human likely didn't hear).
1987
2047
  meetingInterruptContext = `[MEETING — interrupted] You were speaking ("${meetingAgentSpeakingText.slice(0, 140)}") when ${participant} started talking and cut you off. They likely didn't hear the rest. When you respond, briefly acknowledge and adapt — don't just repeat.`;
1988
- meetingAgentSpeaking = false;
1989
- if (meetingSpeakClearTimer)
1990
- clearTimeout(meetingSpeakClearTimer);
2048
+ // Actually STOP the voice: canvas stop + Recall output_audio stop +
2049
+ // queue-generation bump (drops anything still queued). Before 0.9.121
2050
+ // only the canvas was stopped — the real output_audio kept playing.
2051
+ pushCanvas({ kind: 'stop' });
2052
+ interruptMeetingSpeech(`human ${participant} barged in`);
1991
2053
  }
1992
2054
  }
1993
2055
  else {
@@ -1998,7 +2060,15 @@ async function main() {
1998
2060
  // (user directive 2026-08-01).
1999
2061
  if (!isBot && meetingTranscriptBuffer.length) {
2000
2062
  const latchOpen = Date.now() < meetingAddressedUntil || meetingSpeakers.size <= 1;
2001
- flushMeetingBuffer(botId, latchOpen);
2063
+ if (latchOpen) {
2064
+ // Hard silence boundary → the speaker really finished. Flush sooner
2065
+ // than the transcript debounce (still debounced so back-to-back
2066
+ // speakers coalesce into one turn).
2067
+ scheduleAddressedFlush(botId, 450);
2068
+ }
2069
+ else {
2070
+ flushMeetingBuffer(botId, false); // silent observer note-taking batch
2071
+ }
2002
2072
  }
2003
2073
  }
2004
2074
  });
@@ -5257,19 +5327,22 @@ async function main() {
5257
5327
  recallJoin.registerBot(botId, sessionId);
5258
5328
  activeMeetingBotId = botId;
5259
5329
  await sendToFrontend({ type: 'meeting_joined', botId, message: 'Osborn has joined the meeting' });
5260
- // System injection so the LLM knows it's in a meeting and which
5261
- // skill to apply. The meetings skill (agent/.claude/skills/meetings/SKILL.md)
5262
- // teaches the agent: don't speak in response to [MEETING — *]:
5263
- // messages, keep meeting-todos.md updated in the workspace, etc.
5330
+ // Minimal awareness injection (user directive 2026-08-04): tell the
5331
+ // LLM it's in a meeting and how transcripts are tagged — nothing
5332
+ // more. NO "do NOT speak / silent observer" coaching (that fought
5333
+ // the goal of the reply being exactly the main agent's response).
5334
+ // The agent responds to meeting turns the same way it responds on
5335
+ // the website; note-taking is an optional background task, not a
5336
+ // replacement for responding.
5264
5337
  if (currentLLM) {
5265
5338
  try {
5266
5339
  const sysCtx = new llm.ChatContext();
5267
5340
  sysCtx.addMessage({
5268
5341
  role: 'user',
5269
- content: `[SYSTEM] You are now in a meeting (Recall bot ID: ${botId}, URL: ${meetingUrl}). Transcript chunks will arrive every ~30 seconds tagged \`[MEETING — ${botId}]:\`. Follow the meetings skill: do NOT speak in response (no TTS output), instead maintain meeting-todos.md in the session workspace, optionally trigger background research silently. The voice-native user can still interact normally — only the meeting-tagged messages are the silent-observer path. Acknowledge by writing the initial meeting-todos.md skeleton.`,
5342
+ content: `[SYSTEM] You are now in a meeting (Recall bot ID: ${botId}, URL: ${meetingUrl}). Live transcript arrives tagged \`[MEETING — ${botId}]:\`. Respond to what's said exactly as you naturally would — same as on the website, no special meeting phrasing. You may keep meeting-todos.md updated in the workspace in the background, but responding comes first.`,
5270
5343
  });
5271
5344
  currentLLM.chat({ chatCtx: sysCtx });
5272
- console.log('📓 Meeting system injection sent to LLM');
5345
+ console.log('📓 Meeting awareness injection sent to LLM');
5273
5346
  }
5274
5347
  catch (sysErr) {
5275
5348
  console.warn('⚠️ Meeting system injection failed:', sysErr.message);
@@ -110,6 +110,14 @@ export declare class RecallClient extends EventEmitter {
110
110
  * was "barely bearable"; this is the direct loud path.)
111
111
  */
112
112
  outputAudio(botId: string, mp3: Buffer): Promise<boolean>;
113
+ /**
114
+ * Stop the bot's currently-playing output_audio (barge-in / interruption).
115
+ * Recall exposes DELETE on the same endpoint to clear in-progress playback.
116
+ * Best-effort: returns true on 2xx, false otherwise (older bots / no active
117
+ * audio may 404 — harmless, the speech queue's generation bump still halts
118
+ * anything queued). Mirrors the website path's TTS interrupt on barge-in.
119
+ */
120
+ stopOutputAudio(botId: string): Promise<boolean>;
113
121
  getBotStatus(botId: string): Promise<string>;
114
122
  handleWebhook(payload: TranscriptPayload): void;
115
123
  registerBot(botId: string, sessionId: string): void;
@@ -165,6 +165,31 @@ export class RecallClient extends EventEmitter {
165
165
  }
166
166
  return true;
167
167
  }
168
+ /**
169
+ * Stop the bot's currently-playing output_audio (barge-in / interruption).
170
+ * Recall exposes DELETE on the same endpoint to clear in-progress playback.
171
+ * Best-effort: returns true on 2xx, false otherwise (older bots / no active
172
+ * audio may 404 — harmless, the speech queue's generation bump still halts
173
+ * anything queued). Mirrors the website path's TTS interrupt on barge-in.
174
+ */
175
+ async stopOutputAudio(botId) {
176
+ try {
177
+ const res = await fetch(`${RECALL_BASE_URL}/bot/${botId}/output_audio/`, {
178
+ method: 'DELETE',
179
+ headers: { 'Authorization': `Token ${this.#apiKey}` },
180
+ });
181
+ if (!res.ok) {
182
+ const e = await res.text().catch(() => '');
183
+ console.warn(`⚠️ Recall stop output_audio ${res.status}: ${e.slice(0, 120)}`);
184
+ return false;
185
+ }
186
+ return true;
187
+ }
188
+ catch (err) {
189
+ console.warn(`⚠️ Recall stop output_audio failed: ${err.message}`);
190
+ return false;
191
+ }
192
+ }
168
193
  async getBotStatus(botId) {
169
194
  const res = await fetch(`${RECALL_BASE_URL}/bot/${botId}`, {
170
195
  headers: { 'Authorization': `Token ${this.#apiKey}` },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.120",
3
+ "version": "0.9.122",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {