osborn 0.9.91 → 0.9.94

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -215,7 +215,28 @@ function pushCanvas(evt) {
215
215
  canvasClients.delete(res);
216
216
  }
217
217
  }
218
- console.log(`🖼️ canvas ${evt.kind}: ${evt.kind === 'say' ? evt.text.slice(0, 60) : evt.mode} → ${canvasClients.size} client(s)`);
218
+ const desc = evt.kind === 'say' ? evt.text.slice(0, 60) : evt.kind === 'show' ? evt.mode : evt.kind;
219
+ console.log(`🖼️ canvas ${evt.kind}: ${desc} → ${canvasClients.size} client(s)`);
220
+ }
221
+ // ── Meeting interruption state (module scope — shared by the /canvas HTTP
222
+ // handler and the recall speech handler in main()) ──────────────────────────
223
+ const meetingBotName = 'Osborn'; // the bot's name in the meeting; ignore its own speech_on
224
+ // True while the bot's TTS is (probably) still playing into the meeting. Set when
225
+ // a /canvas say is pushed; cleared after an estimated duration OR on interruption.
226
+ // A HUMAN's speech_on while this is true → interrupt.
227
+ let meetingAgentSpeaking = false;
228
+ let meetingSpeakClearTimer = null;
229
+ let meetingAgentSpeakingText = '';
230
+ // Prepended to the next flush: what the bot was cut off saying + who interrupted
231
+ // (same pattern as voice-native interruptions).
232
+ let meetingInterruptContext = '';
233
+ function markMeetingSpeaking(text) {
234
+ meetingAgentSpeaking = true;
235
+ meetingAgentSpeakingText = text;
236
+ if (meetingSpeakClearTimer)
237
+ clearTimeout(meetingSpeakClearTimer);
238
+ const ms = Math.min(30_000, 3_000 + (text.split(/\s+/).length / 2.5) * 1000); // ~2.5 wps + ~3s Recall lag
239
+ meetingSpeakClearTimer = setTimeout(() => { meetingAgentSpeaking = false; meetingAgentSpeakingText = ''; }, ms);
219
240
  }
220
241
  function startApiServer(workingDir, port) {
221
242
  const server = createServer(async (req, res) => {
@@ -558,9 +579,12 @@ function startApiServer(workingDir, port) {
558
579
  req.on('end', () => {
559
580
  try {
560
581
  const evt = JSON.parse(body || '{}');
561
- if (evt.kind !== 'say' && evt.kind !== 'show')
562
- throw new Error("kind must be 'say' or 'show'");
582
+ if (evt.kind !== 'say' && evt.kind !== 'show' && evt.kind !== 'stop')
583
+ throw new Error("kind must be 'say', 'show', or 'stop'");
563
584
  pushCanvas(evt);
585
+ // Track that the bot is now speaking into the meeting → enables interruption.
586
+ if (evt.kind === 'say')
587
+ markMeetingSpeaking(evt.text);
564
588
  res.writeHead(200, { 'Content-Type': 'application/json' });
565
589
  res.end(JSON.stringify({ ok: true, clients: canvasClients.size }));
566
590
  }
@@ -1559,24 +1583,33 @@ async function main() {
1559
1583
  // LIVE meeting transcript → LLM (buffered webhook finals). See recall.on('transcript').
1560
1584
  const meetingTranscriptBuffer = [];
1561
1585
  let meetingFlushTimer = null;
1586
+ // Flush the buffered meeting turns to the LLM. `addressed=true` (the agent was
1587
+ // called by name / asked directly) tags the message so the agent RESPONDS by
1588
+ // speaking into the meeting; otherwise it's a silent-observer note-taking batch.
1589
+ const flushMeetingBuffer = (botId, addressed = false) => {
1590
+ if (!meetingTranscriptBuffer.length || !currentLLM)
1591
+ return;
1592
+ const turns = meetingTranscriptBuffer.splice(0); // drain
1593
+ const header = addressed
1594
+ ? `[MEETING — ${botId}] — YOU WERE ADDRESSED. Respond OUT LOUD into the meeting now: POST http://localhost:${apiPort}/canvas {"kind":"say","text":"..."} with a short, direct reply. Then note it.`
1595
+ : `[MEETING — ${botId}]:`;
1596
+ // Prepend + consume any interruption context (bot was cut off mid-sentence).
1597
+ const interrupt = meetingInterruptContext ? `${meetingInterruptContext}\n` : '';
1598
+ meetingInterruptContext = '';
1599
+ try {
1600
+ const ctx = new llm.ChatContext();
1601
+ ctx.addMessage({ role: 'user', content: `${interrupt}${header}\n${turns.join('\n')}` });
1602
+ currentLLM.chat({ chatCtx: ctx });
1603
+ console.log(`📓 Flushed ${turns.length} meeting turn(s) to LLM${addressed ? ' (ADDRESSED — respond)' : ''}`);
1604
+ }
1605
+ catch (err) {
1606
+ console.warn(`⚠️ Meeting flush failed: ${err.message}`);
1607
+ }
1608
+ };
1562
1609
  const startMeetingFlush = (botId) => {
1563
1610
  stopMeetingFlush();
1564
1611
  console.log(`📓 Meeting transcript flush timer started (bot ${botId}, 20s)`);
1565
- meetingFlushTimer = setInterval(() => {
1566
- if (!meetingTranscriptBuffer.length || !currentLLM)
1567
- return;
1568
- const turns = meetingTranscriptBuffer.splice(0); // drain
1569
- const tagged = `[MEETING — ${botId}]:\n${turns.join('\n')}`;
1570
- try {
1571
- const ctx = new llm.ChatContext();
1572
- ctx.addMessage({ role: 'user', content: tagged });
1573
- currentLLM.chat({ chatCtx: ctx });
1574
- console.log(`📓 Flushed ${turns.length} meeting turn(s) to LLM`);
1575
- }
1576
- catch (err) {
1577
- console.warn(`⚠️ Meeting flush failed: ${err.message}`);
1578
- }
1579
- }, 20_000);
1612
+ meetingFlushTimer = setInterval(() => flushMeetingBuffer(botId, false), 20_000);
1580
1613
  };
1581
1614
  const stopMeetingFlush = () => {
1582
1615
  if (meetingFlushTimer) {
@@ -1633,8 +1666,40 @@ async function main() {
1633
1666
  // is empty until the meeting ENDS, so mid-call the LLM saw nothing.
1634
1667
  recall.on('transcript', ({ botId, speaker, text, partial }) => {
1635
1668
  console.log(`📝 Meeting transcript [${speaker}]${partial ? ' (partial)' : ''}: ${text}`);
1636
- if (!partial && text.trim())
1669
+ if (!partial && text.trim()) {
1637
1670
  meetingTranscriptBuffer.push(`${speaker}: ${text.trim()}`);
1671
+ // Addressed by name → flush NOW (don't wait the 20s batch) and tag it so
1672
+ // the agent replies out loud into the meeting. This is the chat-mode path:
1673
+ // named/asked → prompt response. Un-addressed chunks stay in the silent
1674
+ // 20s batch for note-taking. That's the two-mode seam, mechanically.
1675
+ if (/\bosborn\b/i.test(text)) {
1676
+ console.log('📓 Addressed by name — immediate flush for a response');
1677
+ flushMeetingBuffer(botId, true);
1678
+ }
1679
+ }
1680
+ });
1681
+ // Participant speech VAD → interruption + silence-boundary chunking.
1682
+ recall.on('speech', ({ botId, participant, active }) => {
1683
+ const isBot = participant === meetingBotName || participant === 'Osborn' || participant === 'Unknown';
1684
+ if (active) {
1685
+ // A HUMAN started talking. If the bot is mid-speech → INTERRUPT: stop the
1686
+ // bot's audio and record what it was cut off saying so the next flush can
1687
+ // tell it what it missed (same pattern as voice-native interruptions).
1688
+ if (meetingAgentSpeaking && !isBot) {
1689
+ console.log(`✋ Interruption — ${participant} spoke while bot was talking. Stopping bot audio.`);
1690
+ pushCanvas({ kind: 'stop' });
1691
+ meetingInterruptContext = `[MEETING — interrupted] You were speaking ("${meetingAgentSpeakingText.slice(0, 140)}") when ${participant} started talking and cut you off. They likely didn't hear the rest. When you respond, briefly acknowledge and adapt — don't just repeat.`;
1692
+ meetingAgentSpeaking = false;
1693
+ if (meetingSpeakClearTimer)
1694
+ clearTimeout(meetingSpeakClearTimer);
1695
+ }
1696
+ }
1697
+ else {
1698
+ // speech_off — a natural silence boundary. Flush the buffered turns now
1699
+ // (chunk on conversational pauses, not an arbitrary clock), unless empty.
1700
+ if (!isBot && meetingTranscriptBuffer.length)
1701
+ flushMeetingBuffer(botId, false);
1702
+ }
1638
1703
  });
1639
1704
  }
1640
1705
  // ============================================================
@@ -2287,6 +2352,14 @@ async function main() {
2287
2352
  console.log(`🔇 tts_say fired but text is empty — skipping`);
2288
2353
  return;
2289
2354
  }
2355
+ // Suppress browser TTS for meeting-chunk responses — the agent speaks INTO
2356
+ // the meeting via /canvas, not the laptop speakers (which the Meet mic
2357
+ // re-captures in the same room → feedback). Set by PipelineDirectLLM.chat()
2358
+ // when the turn is a [MEETING —] chunk. Normal user turns are unaffected.
2359
+ if (directLLM.suppressMeetingTTS) {
2360
+ console.log(`🔇 tts_say suppressed (meeting turn — response goes to /canvas, not browser): "${data.text.slice(0, 60)}"`);
2361
+ return;
2362
+ }
2290
2363
  // Suppress while the user is mid-utterance. Without this, agent text generated
2291
2364
  // in parallel by the Claude SDK plays right over the user — same problem as
2292
2365
  // pre-interrupt overlap, but at the *output* side. The suppressed text gets
@@ -42,6 +42,7 @@ export interface FastBrainPanelResult {
42
42
  }
43
43
  export declare class PipelineDirectLLM extends llm.LLM {
44
44
  #private;
45
+ suppressMeetingTTS: boolean;
45
46
  constructor(opts: PipelineDirectOptions);
46
47
  /** Stop the index watcher (call on disconnect/session switch) */
47
48
  stopIndexWatcher(): void;
@@ -18,6 +18,10 @@ export class PipelineDirectLLM extends llm.LLM {
18
18
  #turnAbort = null;
19
19
  #indexWatcher = null;
20
20
  #indexBuilding = false;
21
+ // True while the in-flight turn is a [MEETING —] chunk — the tts_say gate in
22
+ // index.ts reads this to skip browser session.say() (meeting audio goes via
23
+ // /canvas, not the laptop speakers → no same-room feedback). Set per chat().
24
+ suppressMeetingTTS = false;
21
25
  constructor(opts) {
22
26
  super();
23
27
  this.#claudeLLM = new ClaudeLLM(opts);
@@ -175,10 +179,20 @@ export class PipelineDirectLLM extends llm.LLM {
175
179
  }
176
180
  }
177
181
  }
182
+ // Meeting chunks ([MEETING —]) are the silent-observer / addressed-response
183
+ // path. They must NOT drive the browser voice session: (a) no session.say()
184
+ // TTS — the Meet mic re-captures browser audio in the same room → feedback;
185
+ // the agent speaks INTO the meeting via /canvas instead; (b) no fast brain —
186
+ // it's chat-panel noise ("Is there something you'd like to know?") on meeting
187
+ // speech. Set BEFORE the response streams so the tts_say gate (index.ts) sees
188
+ // it. Cleared implicitly by the next real user turn (which isn't a [MEETING]).
189
+ const isMeetingChunk = userText.startsWith('[MEETING');
190
+ this.suppressMeetingTTS = isMeetingChunk;
178
191
  // Fire Claude
179
192
  const claudeStream = this.#claudeLLM.chat({ chatCtx, toolCtx, connOptions, abortController });
180
- // Fire pipeline fast brain in background — no await, no blocking
181
- if (userText.trim()) {
193
+ // Fire pipeline fast brain in background — no await, no blocking. Skip for
194
+ // meeting chunks (silent observer / meeting-response path, not a user turn).
195
+ if (userText.trim() && !isMeetingChunk) {
182
196
  this.#firePipelineFastBrain(userText);
183
197
  }
184
198
  return claudeStream;
@@ -45,8 +45,12 @@ export class RecallClient extends EventEmitter {
45
45
  // transcript turn to /webhook/recall as it's spoken — the receiver
46
46
  // (handleWebhook) already exists. Polling stays as the after-the-fact
47
47
  // backstop. Only wire the webhook when we have a public URL to receive on.
48
+ // Subscribe to transcript (content) AND participant speech VAD events. The
49
+ // speech_on/speech_off events fire from raw audio (faster than transcription)
50
+ // and tell us WHO is speaking WHEN — used for interruption (human speaks while
51
+ // the bot is talking → stop) and for chunking on natural silence boundaries.
48
52
  const realtime = /^https:\/\//.test(webhookBaseUrl)
49
- ? [{ type: 'webhook', url: `${webhookBaseUrl}/webhook/recall`, events: ['transcript.data', 'transcript.partial_data'] }]
53
+ ? [{ type: 'webhook', url: `${webhookBaseUrl}/webhook/recall`, events: ['transcript.data', 'transcript.partial_data', 'participant_events.speech_on', 'participant_events.speech_off'] }]
50
54
  : [];
51
55
  if (!realtime.length)
52
56
  console.log('⚠️ Recall realtime webhook skipped (no public https URL) — polling only');
@@ -154,6 +158,21 @@ export class RecallClient extends EventEmitter {
154
158
  // grows meaningfully or finalizes.
155
159
  #lastEmitted = new Map();
156
160
  handleWebhook(payload) {
161
+ // Participant speech VAD (speech_on/speech_off) — fires from raw audio, ahead
162
+ // of transcription. Emit a 'speech' event so index.ts can (a) interrupt the
163
+ // bot when a HUMAN starts talking over it, and (b) chunk the transcript on
164
+ // natural silence boundaries. The bot's own output_media audio may also fire
165
+ // speech_on for its own participant — index.ts filters that by name.
166
+ if (payload.event === 'participant_events.speech_on' || payload.event === 'participant_events.speech_off') {
167
+ const p = payload.data;
168
+ this.emit('speech', {
169
+ botId: p?.bot?.id ?? 'unknown',
170
+ participant: p?.data?.participant?.name ?? 'Unknown',
171
+ isHost: !!p?.data?.participant?.is_host,
172
+ active: payload.event === 'participant_events.speech_on',
173
+ });
174
+ return;
175
+ }
157
176
  // Accept BOTH finals (transcript.data) AND partials (transcript.partial_data).
158
177
  // recallai_streaming in prioritize_low_latency mode emits partials DURING
159
178
  // the call and finals lag — ignoring partials (the old behavior) meant no
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.91",
3
+ "version": "0.9.94",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {