osborn 0.9.90 → 0.9.93

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -39,6 +39,35 @@ Task(
39
39
 
40
40
  Give the writer everything it needs in the prompt: the session-workspace path, the bot ID, the `us-west-2.recall.ai` endpoint rule, and the `meeting-todos.md` structure. The writer runs the curl/jq/Write steps itself. For research, delegate to the `researcher` sub-agent the same way.
41
41
 
42
+ ## How to SPEAK INTO the meeting (out of silent mode)
43
+
44
+ You can talk directly into the Google Meet / Zoom — your words play as the bot's
45
+ voice. The bot casts a "meeting canvas" webpage as its camera+mic; POSTing to
46
+ your own HTTP API pushes speech (and visuals) to it, and Recall pipes the
47
+ canvas audio into the call. This is ONE Bash call, within budget.
48
+
49
+ **Speak into the meeting:**
50
+ ```bash
51
+ curl -sS -X POST http://localhost:8741/canvas -H 'Content-Type: application/json' \
52
+ -d '{"kind":"say","text":"YOUR WORDS HERE"}'
53
+ ```
54
+ (Port is `OSBORN_API_PORT`, default 8741.)
55
+
56
+ **Show a visual on the bot's camera** (notes, a link, a title):
57
+ ```bash
58
+ curl -sS -X POST http://localhost:8741/canvas -H 'Content-Type: application/json' \
59
+ -d '{"kind":"show","mode":"notes","title":"...","items":["...","..."]}'
60
+ ```
61
+ `mode` = `idle` | `notes` (title+items) | `link` (url) | `web` (iframe url) | `text` (title+text).
62
+
63
+ **When to speak into the meeting:** By DEFAULT stay silent (observer) for
64
+ `[MEETING — *]:` chunks — take notes, don't interrupt. Speak into the meeting
65
+ ONLY when: (a) the voice-native user explicitly tells you to say something to the
66
+ meeting / "tell them X" / "answer that", or (b) you're directly addressed by name
67
+ in the meeting and the user has enabled active mode. When you do speak, keep it
68
+ short and let the room continue. This is the toggle between silent-observer and
69
+ active-participant.
70
+
42
71
  ## How to behave (auto-tagged chunks)
43
72
 
44
73
  For every `[MEETING — *]:` message:
@@ -164,7 +164,14 @@ PKG_SKILLS_DIR="/usr/local/lib/node_modules/osborn/.claude/skills"
164
164
  SEED_VERSION_FILE="${HOME_SKILLS_DIR}/.seed-version"
165
165
  mkdir -p "$HOME_SKILLS_DIR"
166
166
  CURRENT_SEED_VERSION=$(cat "$SEED_VERSION_FILE" 2>/dev/null | tr -d '[:space:]' || echo "")
167
- IMAGE_SEED_VERSION="${OSBORN_IMAGE_VERSION:-latest}"
167
+ # Seed version = the INSTALLED package's own version (reliable on every boot),
168
+ # NOT the OSBORN_IMAGE_VERSION env — that env isn't threaded through
169
+ # `fly machine update`, so it went stale and left the marker permanently pinned
170
+ # at 0.9.48, meaning the refresh below never fired and new-version skills never
171
+ # reached the load path. Reading package.json makes CURRENT != IMAGE trip on a
172
+ # real version bump. Falls back to the env / "latest" if the package is unreadable.
173
+ IMAGE_SEED_VERSION=$(grep -m1 '"version"' /usr/local/lib/node_modules/osborn/package.json 2>/dev/null | sed 's/.*"version"[^"]*"\([^"]*\)".*/\1/')
174
+ [ -z "$IMAGE_SEED_VERSION" ] && IMAGE_SEED_VERSION="${OSBORN_IMAGE_VERSION:-latest}"
168
175
  if [ -d "$PKG_SKILLS_DIR" ]; then
169
176
  REFRESHED=0
170
177
  SEEDED=0
package/dist/index.js CHANGED
@@ -215,7 +215,28 @@ function pushCanvas(evt) {
215
215
  canvasClients.delete(res);
216
216
  }
217
217
  }
218
- console.log(`🖼️ canvas ${evt.kind}: ${evt.kind === 'say' ? evt.text.slice(0, 60) : evt.mode} → ${canvasClients.size} client(s)`);
218
+ const desc = evt.kind === 'say' ? evt.text.slice(0, 60) : evt.kind === 'show' ? evt.mode : evt.kind;
219
+ console.log(`🖼️ canvas ${evt.kind}: ${desc} → ${canvasClients.size} client(s)`);
220
+ }
221
+ // ── Meeting interruption state (module scope — shared by the /canvas HTTP
222
+ // handler and the recall speech handler in main()) ──────────────────────────
223
+ const meetingBotName = 'Osborn'; // the bot's name in the meeting; ignore its own speech_on
224
+ // True while the bot's TTS is (probably) still playing into the meeting. Set when
225
+ // a /canvas say is pushed; cleared after an estimated duration OR on interruption.
226
+ // A HUMAN's speech_on while this is true → interrupt.
227
+ let meetingAgentSpeaking = false;
228
+ let meetingSpeakClearTimer = null;
229
+ let meetingAgentSpeakingText = '';
230
+ // Prepended to the next flush: what the bot was cut off saying + who interrupted
231
+ // (same pattern as voice-native interruptions).
232
+ let meetingInterruptContext = '';
233
+ function markMeetingSpeaking(text) {
234
+ meetingAgentSpeaking = true;
235
+ meetingAgentSpeakingText = text;
236
+ if (meetingSpeakClearTimer)
237
+ clearTimeout(meetingSpeakClearTimer);
238
+ const ms = Math.min(30_000, 3_000 + (text.split(/\s+/).length / 2.5) * 1000); // ~2.5 wps + ~3s Recall lag
239
+ meetingSpeakClearTimer = setTimeout(() => { meetingAgentSpeaking = false; meetingAgentSpeakingText = ''; }, ms);
219
240
  }
220
241
  function startApiServer(workingDir, port) {
221
242
  const server = createServer(async (req, res) => {
@@ -558,9 +579,12 @@ function startApiServer(workingDir, port) {
558
579
  req.on('end', () => {
559
580
  try {
560
581
  const evt = JSON.parse(body || '{}');
561
- if (evt.kind !== 'say' && evt.kind !== 'show')
562
- throw new Error("kind must be 'say' or 'show'");
582
+ if (evt.kind !== 'say' && evt.kind !== 'show' && evt.kind !== 'stop')
583
+ throw new Error("kind must be 'say', 'show', or 'stop'");
563
584
  pushCanvas(evt);
585
+ // Track that the bot is now speaking into the meeting → enables interruption.
586
+ if (evt.kind === 'say')
587
+ markMeetingSpeaking(evt.text);
564
588
  res.writeHead(200, { 'Content-Type': 'application/json' });
565
589
  res.end(JSON.stringify({ ok: true, clients: canvasClients.size }));
566
590
  }
@@ -1556,6 +1580,45 @@ async function main() {
1556
1580
  let currentUserId = '';
1557
1581
  let activeMeetingBotId = null; // Recall.ai bot ID if in a meeting
1558
1582
  let activeMeetingPoller = null; // Transcript poller bound to that bot
1583
+ // LIVE meeting transcript → LLM (buffered webhook finals). See recall.on('transcript').
1584
+ const meetingTranscriptBuffer = [];
1585
+ let meetingFlushTimer = null;
1586
+ // Flush the buffered meeting turns to the LLM. `addressed=true` (the agent was
1587
+ // called by name / asked directly) tags the message so the agent RESPONDS by
1588
+ // speaking into the meeting; otherwise it's a silent-observer note-taking batch.
1589
+ const flushMeetingBuffer = (botId, addressed = false) => {
1590
+ if (!meetingTranscriptBuffer.length || !currentLLM)
1591
+ return;
1592
+ const turns = meetingTranscriptBuffer.splice(0); // drain
1593
+ const header = addressed
1594
+ ? `[MEETING — ${botId}] — YOU WERE ADDRESSED. Respond OUT LOUD into the meeting now: POST http://localhost:${apiPort}/canvas {"kind":"say","text":"..."} with a short, direct reply. Then note it.`
1595
+ : `[MEETING — ${botId}]:`;
1596
+ // Prepend + consume any interruption context (bot was cut off mid-sentence).
1597
+ const interrupt = meetingInterruptContext ? `${meetingInterruptContext}\n` : '';
1598
+ meetingInterruptContext = '';
1599
+ try {
1600
+ const ctx = new llm.ChatContext();
1601
+ ctx.addMessage({ role: 'user', content: `${interrupt}${header}\n${turns.join('\n')}` });
1602
+ currentLLM.chat({ chatCtx: ctx });
1603
+ console.log(`📓 Flushed ${turns.length} meeting turn(s) to LLM${addressed ? ' (ADDRESSED — respond)' : ''}`);
1604
+ }
1605
+ catch (err) {
1606
+ console.warn(`⚠️ Meeting flush failed: ${err.message}`);
1607
+ }
1608
+ };
1609
+ const startMeetingFlush = (botId) => {
1610
+ stopMeetingFlush();
1611
+ console.log(`📓 Meeting transcript flush timer started (bot ${botId}, 20s)`);
1612
+ meetingFlushTimer = setInterval(() => flushMeetingBuffer(botId, false), 20_000);
1613
+ };
1614
+ const stopMeetingFlush = () => {
1615
+ if (meetingFlushTimer) {
1616
+ clearInterval(meetingFlushTimer);
1617
+ meetingFlushTimer = null;
1618
+ console.log('📓 Meeting flush timer stopped');
1619
+ }
1620
+ meetingTranscriptBuffer.length = 0;
1621
+ };
1559
1622
  // Track the active resume session ID across scopes (ParticipantConnected + DataReceived)
1560
1623
  // Updated by resume_session, session_selected, continue_session, switch_session handlers
1561
1624
  let currentResumeSessionId;
@@ -1593,26 +1656,50 @@ async function main() {
1593
1656
  // "what was said in the meeting" display, separate from the LLM input path).
1594
1657
  const recall = getRecallClient();
1595
1658
  if (recall) {
1596
- console.log('🎥 Recall.ai client initialized (webhook STT receiver — LLM forwarding disabled, see meeting-bot Phase 2)');
1597
- recall.on('transcript', ({ botId, speaker, text }) => {
1598
- console.log(`📝 Meeting transcript [${speaker}]: ${text}`);
1599
- // INTENTIONALLY DISABLED — see comment above. Audio path is now LiveKit
1600
- // → meeting-bot page publishes meeting audio → agent STT processes it.
1601
- // The line below is preserved as a reference for future re-enablement
1602
- // (e.g. as a display-only feature, NOT as LLM input).
1603
- //
1604
- // if (currentLLM && currentSession) {
1605
- // const meetingText = `[Meeting — ${speaker}]: ${text}`
1606
- // try {
1607
- // if (currentVoiceMode === 'pipeline' || currentVoiceMode === 'direct') {
1608
- // const chatCtx = new llm.ChatContext()
1609
- // chatCtx.addMessage({ role: 'user', content: meetingText })
1610
- // ;(currentLLM as any).chat({ chatCtx })
1611
- // }
1612
- // } catch (err) {
1613
- // console.error('❌ Failed to route meeting transcript:', err)
1614
- // }
1615
- // }
1659
+ console.log('🎥 Recall.ai client initialized — LIVE webhook finals buffered → LLM (batch poller download_url is empty mid-call, so the webhook is the only live transcript source)');
1660
+ // LIVE meeting transcript → LLM. The realtime webhook (handleWebhook) emits
1661
+ // every partial + final. We buffer FINALS (partials are too noisy) and a
1662
+ // flush timer (started on join_meeting) batches them to currentLLM.chat()
1663
+ // as [MEETING — botId]: every ~20s — so the agent actually sees the meeting
1664
+ // and can take notes / delegate to the writer. Previously this was DISABLED
1665
+ // and the poller (batch endpoint) was the only LLM path — but that endpoint
1666
+ // is empty until the meeting ENDS, so mid-call the LLM saw nothing.
1667
+ recall.on('transcript', ({ botId, speaker, text, partial }) => {
1668
+ console.log(`📝 Meeting transcript [${speaker}]${partial ? ' (partial)' : ''}: ${text}`);
1669
+ if (!partial && text.trim()) {
1670
+ meetingTranscriptBuffer.push(`${speaker}: ${text.trim()}`);
1671
+ // Addressed by name → flush NOW (don't wait the 20s batch) and tag it so
1672
+ // the agent replies out loud into the meeting. This is the chat-mode path:
1673
+ // named/asked → prompt response. Un-addressed chunks stay in the silent
1674
+ // 20s batch for note-taking. That's the two-mode seam, mechanically.
1675
+ if (/\bosborn\b/i.test(text)) {
1676
+ console.log('📓 Addressed by name — immediate flush for a response');
1677
+ flushMeetingBuffer(botId, true);
1678
+ }
1679
+ }
1680
+ });
1681
+ // Participant speech VAD → interruption + silence-boundary chunking.
1682
+ recall.on('speech', ({ botId, participant, active }) => {
1683
+ const isBot = participant === meetingBotName || participant === 'Osborn' || participant === 'Unknown';
1684
+ if (active) {
1685
+ // A HUMAN started talking. If the bot is mid-speech → INTERRUPT: stop the
1686
+ // bot's audio and record what it was cut off saying so the next flush can
1687
+ // tell it what it missed (same pattern as voice-native interruptions).
1688
+ if (meetingAgentSpeaking && !isBot) {
1689
+ console.log(`✋ Interruption — ${participant} spoke while bot was talking. Stopping bot audio.`);
1690
+ pushCanvas({ kind: 'stop' });
1691
+ meetingInterruptContext = `[MEETING — interrupted] You were speaking ("${meetingAgentSpeakingText.slice(0, 140)}") when ${participant} started talking and cut you off. They likely didn't hear the rest. When you respond, briefly acknowledge and adapt — don't just repeat.`;
1692
+ meetingAgentSpeaking = false;
1693
+ if (meetingSpeakClearTimer)
1694
+ clearTimeout(meetingSpeakClearTimer);
1695
+ }
1696
+ }
1697
+ else {
1698
+ // speech_off — a natural silence boundary. Flush the buffered turns now
1699
+ // (chunk on conversational pauses, not an arbitrary clock), unless empty.
1700
+ if (!isBot && meetingTranscriptBuffer.length)
1701
+ flushMeetingBuffer(botId, false);
1702
+ }
1616
1703
  });
1617
1704
  }
1618
1705
  // ============================================================
@@ -2265,6 +2352,14 @@ async function main() {
2265
2352
  console.log(`🔇 tts_say fired but text is empty — skipping`);
2266
2353
  return;
2267
2354
  }
2355
+ // Suppress browser TTS for meeting-chunk responses — the agent speaks INTO
2356
+ // the meeting via /canvas, not the laptop speakers (which the Meet mic
2357
+ // re-captures in the same room → feedback). Set by PipelineDirectLLM.chat()
2358
+ // when the turn is a [MEETING —] chunk. Normal user turns are unaffected.
2359
+ if (directLLM.suppressMeetingTTS) {
2360
+ console.log(`🔇 tts_say suppressed (meeting turn — response goes to /canvas, not browser): "${data.text.slice(0, 60)}"`);
2361
+ return;
2362
+ }
2268
2363
  // Suppress while the user is mid-utterance. Without this, agent text generated
2269
2364
  // in parallel by the Claude SDK plays right over the user — same problem as
2270
2365
  // pre-interrupt overlap, but at the *output* side. The suppressed text gets
@@ -4101,6 +4196,7 @@ async function main() {
4101
4196
  clearFastBrainSession();
4102
4197
  clearPipelineFastBrainSession();
4103
4198
  // Auto-leave any active meeting bot when user disconnects from the room
4199
+ stopMeetingFlush();
4104
4200
  if (activeMeetingPoller) {
4105
4201
  activeMeetingPoller.stop();
4106
4202
  activeMeetingPoller = null;
@@ -4786,6 +4882,9 @@ async function main() {
4786
4882
  },
4787
4883
  });
4788
4884
  activeMeetingPoller.start();
4885
+ // LIVE path: buffer webhook finals + flush to the LLM every 20s.
4886
+ // (The poller above only lands data after the meeting ENDS.)
4887
+ startMeetingFlush(botId);
4789
4888
  }
4790
4889
  catch (err) {
4791
4890
  console.error('❌ Recall.ai join error:', err);
@@ -4799,8 +4898,9 @@ async function main() {
4799
4898
  const recallLeave = getRecallClient();
4800
4899
  if (recallLeave && botId) {
4801
4900
  try {
4802
- // Stop the transcript poller FIRST so no more transcript chunks get
4803
- // forwarded to the LLM during the leave.
4901
+ // Stop the transcript poller + live flush FIRST so no more chunks
4902
+ // get forwarded to the LLM during the leave.
4903
+ stopMeetingFlush();
4804
4904
  if (activeMeetingPoller) {
4805
4905
  activeMeetingPoller.stop();
4806
4906
  activeMeetingPoller = null;
@@ -42,6 +42,7 @@ export interface FastBrainPanelResult {
42
42
  }
43
43
  export declare class PipelineDirectLLM extends llm.LLM {
44
44
  #private;
45
+ suppressMeetingTTS: boolean;
45
46
  constructor(opts: PipelineDirectOptions);
46
47
  /** Stop the index watcher (call on disconnect/session switch) */
47
48
  stopIndexWatcher(): void;
@@ -18,6 +18,10 @@ export class PipelineDirectLLM extends llm.LLM {
18
18
  #turnAbort = null;
19
19
  #indexWatcher = null;
20
20
  #indexBuilding = false;
21
+ // True while the in-flight turn is a [MEETING —] chunk — the tts_say gate in
22
+ // index.ts reads this to skip browser session.say() (meeting audio goes via
23
+ // /canvas, not the laptop speakers → no same-room feedback). Set per chat().
24
+ suppressMeetingTTS = false;
21
25
  constructor(opts) {
22
26
  super();
23
27
  this.#claudeLLM = new ClaudeLLM(opts);
@@ -175,10 +179,20 @@ export class PipelineDirectLLM extends llm.LLM {
175
179
  }
176
180
  }
177
181
  }
182
+ // Meeting chunks ([MEETING —]) are the silent-observer / addressed-response
183
+ // path. They must NOT drive the browser voice session: (a) no session.say()
184
+ // TTS — the Meet mic re-captures browser audio in the same room → feedback;
185
+ // the agent speaks INTO the meeting via /canvas instead; (b) no fast brain —
186
+ // it's chat-panel noise ("Is there something you'd like to know?") on meeting
187
+ // speech. Set BEFORE the response streams so the tts_say gate (index.ts) sees
188
+ // it. Cleared implicitly by the next real user turn (which isn't a [MEETING]).
189
+ const isMeetingChunk = userText.startsWith('[MEETING');
190
+ this.suppressMeetingTTS = isMeetingChunk;
178
191
  // Fire Claude
179
192
  const claudeStream = this.#claudeLLM.chat({ chatCtx, toolCtx, connOptions, abortController });
180
- // Fire pipeline fast brain in background — no await, no blocking
181
- if (userText.trim()) {
193
+ // Fire pipeline fast brain in background — no await, no blocking. Skip for
194
+ // meeting chunks (silent observer / meeting-response path, not a user turn).
195
+ if (userText.trim() && !isMeetingChunk) {
182
196
  this.#firePipelineFastBrain(userText);
183
197
  }
184
198
  return claudeStream;
@@ -45,8 +45,12 @@ export class RecallClient extends EventEmitter {
45
45
  // transcript turn to /webhook/recall as it's spoken — the receiver
46
46
  // (handleWebhook) already exists. Polling stays as the after-the-fact
47
47
  // backstop. Only wire the webhook when we have a public URL to receive on.
48
+ // Subscribe to transcript (content) AND participant speech VAD events. The
49
+ // speech_on/speech_off events fire from raw audio (faster than transcription)
50
+ // and tell us WHO is speaking WHEN — used for interruption (human speaks while
51
+ // the bot is talking → stop) and for chunking on natural silence boundaries.
48
52
  const realtime = /^https:\/\//.test(webhookBaseUrl)
49
- ? [{ type: 'webhook', url: `${webhookBaseUrl}/webhook/recall`, events: ['transcript.data', 'transcript.partial_data'] }]
53
+ ? [{ type: 'webhook', url: `${webhookBaseUrl}/webhook/recall`, events: ['transcript.data', 'transcript.partial_data', 'participant_events.speech_on', 'participant_events.speech_off'] }]
50
54
  : [];
51
55
  if (!realtime.length)
52
56
  console.log('⚠️ Recall realtime webhook skipped (no public https URL) — polling only');
@@ -154,6 +158,21 @@ export class RecallClient extends EventEmitter {
154
158
  // grows meaningfully or finalizes.
155
159
  #lastEmitted = new Map();
156
160
  handleWebhook(payload) {
161
+ // Participant speech VAD (speech_on/speech_off) — fires from raw audio, ahead
162
+ // of transcription. Emit a 'speech' event so index.ts can (a) interrupt the
163
+ // bot when a HUMAN starts talking over it, and (b) chunk the transcript on
164
+ // natural silence boundaries. The bot's own output_media audio may also fire
165
+ // speech_on for its own participant — index.ts filters that by name.
166
+ if (payload.event === 'participant_events.speech_on' || payload.event === 'participant_events.speech_off') {
167
+ const p = payload.data;
168
+ this.emit('speech', {
169
+ botId: p?.bot?.id ?? 'unknown',
170
+ participant: p?.data?.participant?.name ?? 'Unknown',
171
+ isHost: !!p?.data?.participant?.is_host,
172
+ active: payload.event === 'participant_events.speech_on',
173
+ });
174
+ return;
175
+ }
157
176
  // Accept BOTH finals (transcript.data) AND partials (transcript.partial_data).
158
177
  // recallai_streaming in prioritize_low_latency mode emits partials DURING
159
178
  // the call and finals lag — ignoring partials (the old behavior) meant no
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.90",
3
+ "version": "0.9.93",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {