osborn 0.9.120 → 0.9.121

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -235,6 +235,38 @@ let meetingAgentSpeakingText = '';
235
235
  // Prepended to the next flush: what the bot was cut off saying + who interrupted
236
236
  // (same pattern as voice-native interruptions).
237
237
  let meetingInterruptContext = '';
238
+ // Meeting speech QUEUE (0.9.121): serialize output_audio so replies never
239
+ // overlap — the Recall-sink equivalent of session.say's SpeechHandle queue.
240
+ // Recall's output_audio POST returns on ACCEPT, not on finish, so without this
241
+ // two replies (from two flushes, or a streamed multi-chunk reply) play ON TOP
242
+ // of each other — the "another voice over it" the user heard. Each utterance
243
+ // waits for the prior one's estimated playback before it plays; a generation
244
+ // counter (bumped on human interruption / a superseding turn) discards anything
245
+ // still queued so the bot never talks over itself or a human.
246
+ let meetingSpeakChain = Promise.resolve();
247
+ let meetingSpeakGen = 0;
248
+ const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
249
+ // Estimated playback duration of a spoken line: ~2.5 words/sec + ~0.8s Recall
250
+ // buffer. Used to hold the speech queue so the next utterance doesn't overlap.
251
+ function estimatedSpeechMs(text) {
252
+ const words = text.split(/\s+/).filter(Boolean).length;
253
+ return Math.min(30_000, 800 + (words / 2.5) * 1000);
254
+ }
255
+ // Interrupt all meeting speech: bump the generation (drops queued + in-synth
256
+ // utterances) and stop any output_audio Recall is currently playing.
257
+ function interruptMeetingSpeech(reason) {
258
+ meetingSpeakGen++;
259
+ meetingAgentSpeaking = false;
260
+ if (meetingSpeakClearTimer) {
261
+ clearTimeout(meetingSpeakClearTimer);
262
+ meetingSpeakClearTimer = null;
263
+ }
264
+ const recall = getRecallClient();
265
+ const botId = recall?.getActiveBotIds?.()[0];
266
+ if (recall && botId)
267
+ void recall.stopOutputAudio(botId);
268
+ console.log(`✋ meeting speech interrupted (${reason}) — queue cleared + output_audio stopped`);
269
+ }
238
270
  // Synthesize speech as MP3 (Deepgram fast path, OpenAI fallback) — for
239
271
  // Recall native output_audio, which requires mp3.
240
272
  async function synthMp3(text) {
@@ -282,21 +314,43 @@ async function synthMp3(text) {
282
314
  // bot's camera shows what it's saying — no double-audio. output_audio +
283
315
  // canvas camera coexist (confirmed: user heard output_audio while the canvas
284
316
  // was showing). Falls back to canvas 'say' (audio) only if output_audio fails.
285
- async function speakIntoMeeting(text) {
286
- const recall = getRecallClient();
287
- const botId = recall?.getActiveBotIds?.()[0];
288
- if (recall && botId) {
289
- const mp3 = await synthMp3(text);
290
- if (mp3 && await recall.outputAudio(botId, mp3)) {
291
- console.log(`📢 spoke via Recall output_audio (${mp3.length}b): "${text.slice(0, 60)}"`);
292
- pushCanvas({ kind: 'caption', text }); // visual only, no audio
293
- markMeetingSpeaking(text);
317
+ function speakIntoMeeting(text) {
318
+ if (!text?.trim())
319
+ return Promise.resolve();
320
+ // Capture the generation at ENQUEUE time. If an interrupt (or a superseding
321
+ // turn) bumps the gen before this item runs — or mid-synth — we drop it, so
322
+ // the bot never plays a reply the conversation has already moved past.
323
+ const gen = meetingSpeakGen;
324
+ const run = meetingSpeakChain.then(async () => {
325
+ if (gen !== meetingSpeakGen) {
326
+ console.log(`🔇 meeting speech superseded — dropping: "${text.slice(0, 40)}"`);
294
327
  return;
295
328
  }
296
- }
297
- console.log(`📽️ falling back to canvas say (audio): "${text.slice(0, 50)}"`);
298
- pushCanvas({ kind: 'say', text });
299
- markMeetingSpeaking(text);
329
+ const recall = getRecallClient();
330
+ const botId = recall?.getActiveBotIds?.()[0];
331
+ if (recall && botId) {
332
+ const mp3 = await synthMp3(text);
333
+ if (gen !== meetingSpeakGen) {
334
+ console.log(`🔇 meeting speech interrupted mid-synth — dropping: "${text.slice(0, 40)}"`);
335
+ return;
336
+ }
337
+ if (mp3 && await recall.outputAudio(botId, mp3)) {
338
+ console.log(`📢 spoke via Recall output_audio (${mp3.length}b): "${text.slice(0, 60)}"`);
339
+ pushCanvas({ kind: 'caption', text }); // visual only, no audio
340
+ markMeetingSpeaking(text);
341
+ // Hold the queue for the estimated playback so the NEXT utterance
342
+ // doesn't start on top of this one (POST returns on accept, not finish).
343
+ await sleep(estimatedSpeechMs(text));
344
+ return;
345
+ }
346
+ }
347
+ console.log(`📽️ falling back to canvas say (audio): "${text.slice(0, 50)}"`);
348
+ pushCanvas({ kind: 'say', text });
349
+ markMeetingSpeaking(text);
350
+ await sleep(estimatedSpeechMs(text));
351
+ }).catch((e) => { console.warn(`⚠️ meeting speak failed: ${e.message}`); });
352
+ meetingSpeakChain = run;
353
+ return run;
300
354
  }
301
355
  function markMeetingSpeaking(text) {
302
356
  meetingAgentSpeaking = true;
@@ -1775,12 +1829,12 @@ async function main() {
1775
1829
  // from muting the audio path. Reset by endMeeting.
1776
1830
  meetingAddressedUntil = Date.now() + 6 * 60 * 60 * 1000;
1777
1831
  }
1778
- // PROMPT PARITY (user directive 2026-08-01): addressed turns carry ONLY a
1779
- // minimal tag — no behavioral re-instruction. The agent replies exactly as
1780
- // it would to a regular voice turn; the tts_say→canvas redirect handles
1781
- // where the words go. Same agent, same behavior, both fronts.
1832
+ // Addressed turns get a MINIMAL tag + a hard brevity rule (0.9.121): the
1833
+ // reply is SPOKEN into a live meeting, so long answers (a) take 8s+ to
1834
+ // synthesize and (b) pile up / talk over the next turn. 1–2 sentences keeps
1835
+ // it conversational and fast; the agent can offer to go deeper if asked.
1782
1836
  const header = addressed
1783
- ? `[MEETING — ${botId}] (addressed — reply is spoken into the meeting):`
1837
+ ? `[MEETING — ${botId}] (addressed — your reply is SPOKEN OUT LOUD into the meeting: keep it to 1–2 short sentences, conversational, no lists/markdown; offer to elaborate only if they want more):`
1784
1838
  : `[MEETING — ${botId}]:`;
1785
1839
  // Prepend + consume any interruption context (bot was cut off mid-sentence).
1786
1840
  const interrupt = meetingInterruptContext ? `${meetingInterruptContext}\n` : '';
@@ -1806,8 +1860,33 @@ async function main() {
1806
1860
  meetingFlushTimer = null;
1807
1861
  console.log('📓 Meeting flush timer stopped');
1808
1862
  }
1863
+ if (addressedFlushTimer) {
1864
+ clearTimeout(addressedFlushTimer);
1865
+ addressedFlushTimer = null;
1866
+ }
1809
1867
  meetingTranscriptBuffer.length = 0;
1810
1868
  };
1869
+ // TURN-DEBOUNCED addressed flush (0.9.121): Recall closes a transcript
1870
+ // segment on every pause, so flushing on each final made the bot reply to
1871
+ // FRAGMENTS mid-thought (and multiple times per utterance). Instead we
1872
+ // debounce: each new transcript final resets a short timer; we only flush
1873
+ // (= reply) after the speaker has actually paused, so one turn = one reply.
1874
+ // speech_off (a hard silence boundary) can flush sooner via a smaller delay.
1875
+ let addressedFlushTimer = null;
1876
+ const ADDRESSED_DEBOUNCE_MS = 1400;
1877
+ const scheduleAddressedFlush = (botId, delayMs = ADDRESSED_DEBOUNCE_MS) => {
1878
+ if (addressedFlushTimer)
1879
+ clearTimeout(addressedFlushTimer);
1880
+ addressedFlushTimer = setTimeout(() => {
1881
+ addressedFlushTimer = null;
1882
+ if (!meetingTranscriptBuffer.length)
1883
+ return;
1884
+ // A fresh human turn supersedes anything the bot was still saying about
1885
+ // the previous one — interrupt stale queued/playing speech before we reply.
1886
+ interruptMeetingSpeech('new addressed turn');
1887
+ flushMeetingBuffer(botId, true);
1888
+ }, delayMs);
1889
+ };
1811
1890
  // ── Meeting lifecycle (centralized teardown, 0.9.95) ──
1812
1891
  // The bot's lifecycle FOLLOWS the voice session (deliberate coupling — a
1813
1892
  // decoupled always-on bot means untracked background agents; revisit only
@@ -1969,8 +2048,12 @@ async function main() {
1969
2048
  }
1970
2049
  const oneOnOne = meetingSpeakers.size <= 1;
1971
2050
  if (oneOnOne || /\b(osborne?|oz\s?born|os\s?born|was born|is born|ozborn|osbourne?|austin\b.{0,8}(hear|there|can you))/i.test(text)) {
1972
- console.log(`📓 Addressed (${oneOnOne ? '1:1 meeting' : 'by name'}) — immediate flush for a response`);
1973
- flushMeetingBuffer(botId, true);
2051
+ // DEBOUNCED (0.9.121): don't reply to this fragment — wait for the
2052
+ // speaker to actually pause. Each new final resets the timer, so one
2053
+ // continuous thought (even across Recall's mid-sentence segment splits)
2054
+ // becomes ONE reply instead of several talking over each other.
2055
+ console.log(`📓 Addressed (${oneOnOne ? '1:1 meeting' : 'by name'}) — turn debounced (~${ADDRESSED_DEBOUNCE_MS}ms)`);
2056
+ scheduleAddressedFlush(botId);
1974
2057
  }
1975
2058
  }
1976
2059
  });
@@ -1982,12 +2065,16 @@ async function main() {
1982
2065
  // bot's audio and record what it was cut off saying so the next flush can
1983
2066
  // tell it what it missed (same pattern as voice-native interruptions).
1984
2067
  if (meetingAgentSpeaking && !isBot) {
1985
- console.log(`✋ Interruption — ${participant} spoke while bot was talking. Stopping bot audio.`);
1986
- pushCanvas({ kind: 'stop' });
2068
+ console.log(`✋ Interruption — ${participant} spoke while bot was talking.`);
2069
+ // Capture what got cut off BEFORE interrupting (same interruption-
2070
+ // context ledger as the website path — the next flush tells the agent
2071
+ // it was cut off + what the human likely didn't hear).
1987
2072
  meetingInterruptContext = `[MEETING — interrupted] You were speaking ("${meetingAgentSpeakingText.slice(0, 140)}") when ${participant} started talking and cut you off. They likely didn't hear the rest. When you respond, briefly acknowledge and adapt — don't just repeat.`;
1988
- meetingAgentSpeaking = false;
1989
- if (meetingSpeakClearTimer)
1990
- clearTimeout(meetingSpeakClearTimer);
2073
+ // Actually STOP the voice: canvas stop + Recall output_audio stop +
2074
+ // queue-generation bump (drops anything still queued). Before 0.9.121
2075
+ // only the canvas was stopped — the real output_audio kept playing.
2076
+ pushCanvas({ kind: 'stop' });
2077
+ interruptMeetingSpeech(`human ${participant} barged in`);
1991
2078
  }
1992
2079
  }
1993
2080
  else {
@@ -1998,7 +2085,15 @@ async function main() {
1998
2085
  // (user directive 2026-08-01).
1999
2086
  if (!isBot && meetingTranscriptBuffer.length) {
2000
2087
  const latchOpen = Date.now() < meetingAddressedUntil || meetingSpeakers.size <= 1;
2001
- flushMeetingBuffer(botId, latchOpen);
2088
+ if (latchOpen) {
2089
+ // Hard silence boundary → the speaker really finished. Flush sooner
2090
+ // than the transcript debounce (still debounced so back-to-back
2091
+ // speakers coalesce into one turn).
2092
+ scheduleAddressedFlush(botId, 450);
2093
+ }
2094
+ else {
2095
+ flushMeetingBuffer(botId, false); // silent observer note-taking batch
2096
+ }
2002
2097
  }
2003
2098
  }
2004
2099
  });
@@ -110,6 +110,14 @@ export declare class RecallClient extends EventEmitter {
110
110
  * was "barely bearable"; this is the direct loud path.)
111
111
  */
112
112
  outputAudio(botId: string, mp3: Buffer): Promise<boolean>;
113
+ /**
114
+ * Stop the bot's currently-playing output_audio (barge-in / interruption).
115
+ * Recall exposes DELETE on the same endpoint to clear in-progress playback.
116
+ * Best-effort: returns true on 2xx, false otherwise (older bots / no active
117
+ * audio may 404 — harmless, the speech queue's generation bump still halts
118
+ * anything queued). Mirrors the website path's TTS interrupt on barge-in.
119
+ */
120
+ stopOutputAudio(botId: string): Promise<boolean>;
113
121
  getBotStatus(botId: string): Promise<string>;
114
122
  handleWebhook(payload: TranscriptPayload): void;
115
123
  registerBot(botId: string, sessionId: string): void;
@@ -165,6 +165,31 @@ export class RecallClient extends EventEmitter {
165
165
  }
166
166
  return true;
167
167
  }
168
+ /**
169
+ * Stop the bot's currently-playing output_audio (barge-in / interruption).
170
+ * Recall exposes DELETE on the same endpoint to clear in-progress playback.
171
+ * Best-effort: returns true on 2xx, false otherwise (older bots / no active
172
+ * audio may 404 — harmless, the speech queue's generation bump still halts
173
+ * anything queued). Mirrors the website path's TTS interrupt on barge-in.
174
+ */
175
+ async stopOutputAudio(botId) {
176
+ try {
177
+ const res = await fetch(`${RECALL_BASE_URL}/bot/${botId}/output_audio/`, {
178
+ method: 'DELETE',
179
+ headers: { 'Authorization': `Token ${this.#apiKey}` },
180
+ });
181
+ if (!res.ok) {
182
+ const e = await res.text().catch(() => '');
183
+ console.warn(`⚠️ Recall stop output_audio ${res.status}: ${e.slice(0, 120)}`);
184
+ return false;
185
+ }
186
+ return true;
187
+ }
188
+ catch (err) {
189
+ console.warn(`⚠️ Recall stop output_audio failed: ${err.message}`);
190
+ return false;
191
+ }
192
+ }
168
193
  async getBotStatus(botId) {
169
194
  const res = await fetch(`${RECALL_BASE_URL}/bot/${botId}`, {
170
195
  headers: { 'Authorization': `Token ${this.#apiKey}` },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.120",
3
+ "version": "0.9.121",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {