osborn 0.9.91 → 0.9.93
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +92 -19
- package/dist/pipeline-direct-llm.d.ts +1 -0
- package/dist/pipeline-direct-llm.js +16 -2
- package/dist/recall-client.js +20 -1
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -215,7 +215,28 @@ function pushCanvas(evt) {
|
|
|
215
215
|
canvasClients.delete(res);
|
|
216
216
|
}
|
|
217
217
|
}
|
|
218
|
-
|
|
218
|
+
const desc = evt.kind === 'say' ? evt.text.slice(0, 60) : evt.kind === 'show' ? evt.mode : evt.kind;
|
|
219
|
+
console.log(`🖼️ canvas ${evt.kind}: ${desc} → ${canvasClients.size} client(s)`);
|
|
220
|
+
}
|
|
221
|
+
// ── Meeting interruption state (module scope — shared by the /canvas HTTP
|
|
222
|
+
// handler and the recall speech handler in main()) ──────────────────────────
|
|
223
|
+
const meetingBotName = 'Osborn'; // the bot's name in the meeting; ignore its own speech_on
|
|
224
|
+
// True while the bot's TTS is (probably) still playing into the meeting. Set when
|
|
225
|
+
// a /canvas say is pushed; cleared after an estimated duration OR on interruption.
|
|
226
|
+
// A HUMAN's speech_on while this is true → interrupt.
|
|
227
|
+
let meetingAgentSpeaking = false;
|
|
228
|
+
let meetingSpeakClearTimer = null;
|
|
229
|
+
let meetingAgentSpeakingText = '';
|
|
230
|
+
// Prepended to the next flush: what the bot was cut off saying + who interrupted
|
|
231
|
+
// (same pattern as voice-native interruptions).
|
|
232
|
+
let meetingInterruptContext = '';
|
|
233
|
+
function markMeetingSpeaking(text) {
|
|
234
|
+
meetingAgentSpeaking = true;
|
|
235
|
+
meetingAgentSpeakingText = text;
|
|
236
|
+
if (meetingSpeakClearTimer)
|
|
237
|
+
clearTimeout(meetingSpeakClearTimer);
|
|
238
|
+
const ms = Math.min(30_000, 3_000 + (text.split(/\s+/).length / 2.5) * 1000); // ~2.5 wps + ~3s Recall lag
|
|
239
|
+
meetingSpeakClearTimer = setTimeout(() => { meetingAgentSpeaking = false; meetingAgentSpeakingText = ''; }, ms);
|
|
219
240
|
}
|
|
220
241
|
function startApiServer(workingDir, port) {
|
|
221
242
|
const server = createServer(async (req, res) => {
|
|
@@ -558,9 +579,12 @@ function startApiServer(workingDir, port) {
|
|
|
558
579
|
req.on('end', () => {
|
|
559
580
|
try {
|
|
560
581
|
const evt = JSON.parse(body || '{}');
|
|
561
|
-
if (evt.kind !== 'say' && evt.kind !== 'show')
|
|
562
|
-
throw new Error("kind must be 'say' or '
|
|
582
|
+
if (evt.kind !== 'say' && evt.kind !== 'show' && evt.kind !== 'stop')
|
|
583
|
+
throw new Error("kind must be 'say', 'show', or 'stop'");
|
|
563
584
|
pushCanvas(evt);
|
|
585
|
+
// Track that the bot is now speaking into the meeting → enables interruption.
|
|
586
|
+
if (evt.kind === 'say')
|
|
587
|
+
markMeetingSpeaking(evt.text);
|
|
564
588
|
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
565
589
|
res.end(JSON.stringify({ ok: true, clients: canvasClients.size }));
|
|
566
590
|
}
|
|
@@ -1559,24 +1583,33 @@ async function main() {
|
|
|
1559
1583
|
// LIVE meeting transcript → LLM (buffered webhook finals). See recall.on('transcript').
|
|
1560
1584
|
const meetingTranscriptBuffer = [];
|
|
1561
1585
|
let meetingFlushTimer = null;
|
|
1586
|
+
// Flush the buffered meeting turns to the LLM. `addressed=true` (the agent was
|
|
1587
|
+
// called by name / asked directly) tags the message so the agent RESPONDS by
|
|
1588
|
+
// speaking into the meeting; otherwise it's a silent-observer note-taking batch.
|
|
1589
|
+
const flushMeetingBuffer = (botId, addressed = false) => {
|
|
1590
|
+
if (!meetingTranscriptBuffer.length || !currentLLM)
|
|
1591
|
+
return;
|
|
1592
|
+
const turns = meetingTranscriptBuffer.splice(0); // drain
|
|
1593
|
+
const header = addressed
|
|
1594
|
+
? `[MEETING — ${botId}] — YOU WERE ADDRESSED. Respond OUT LOUD into the meeting now: POST http://localhost:${apiPort}/canvas {"kind":"say","text":"..."} with a short, direct reply. Then note it.`
|
|
1595
|
+
: `[MEETING — ${botId}]:`;
|
|
1596
|
+
// Prepend + consume any interruption context (bot was cut off mid-sentence).
|
|
1597
|
+
const interrupt = meetingInterruptContext ? `${meetingInterruptContext}\n` : '';
|
|
1598
|
+
meetingInterruptContext = '';
|
|
1599
|
+
try {
|
|
1600
|
+
const ctx = new llm.ChatContext();
|
|
1601
|
+
ctx.addMessage({ role: 'user', content: `${interrupt}${header}\n${turns.join('\n')}` });
|
|
1602
|
+
currentLLM.chat({ chatCtx: ctx });
|
|
1603
|
+
console.log(`📓 Flushed ${turns.length} meeting turn(s) to LLM${addressed ? ' (ADDRESSED — respond)' : ''}`);
|
|
1604
|
+
}
|
|
1605
|
+
catch (err) {
|
|
1606
|
+
console.warn(`⚠️ Meeting flush failed: ${err.message}`);
|
|
1607
|
+
}
|
|
1608
|
+
};
|
|
1562
1609
|
const startMeetingFlush = (botId) => {
|
|
1563
1610
|
stopMeetingFlush();
|
|
1564
1611
|
console.log(`📓 Meeting transcript flush timer started (bot ${botId}, 20s)`);
|
|
1565
|
-
meetingFlushTimer = setInterval(() =>
|
|
1566
|
-
if (!meetingTranscriptBuffer.length || !currentLLM)
|
|
1567
|
-
return;
|
|
1568
|
-
const turns = meetingTranscriptBuffer.splice(0); // drain
|
|
1569
|
-
const tagged = `[MEETING — ${botId}]:\n${turns.join('\n')}`;
|
|
1570
|
-
try {
|
|
1571
|
-
const ctx = new llm.ChatContext();
|
|
1572
|
-
ctx.addMessage({ role: 'user', content: tagged });
|
|
1573
|
-
currentLLM.chat({ chatCtx: ctx });
|
|
1574
|
-
console.log(`📓 Flushed ${turns.length} meeting turn(s) to LLM`);
|
|
1575
|
-
}
|
|
1576
|
-
catch (err) {
|
|
1577
|
-
console.warn(`⚠️ Meeting flush failed: ${err.message}`);
|
|
1578
|
-
}
|
|
1579
|
-
}, 20_000);
|
|
1612
|
+
meetingFlushTimer = setInterval(() => flushMeetingBuffer(botId, false), 20_000);
|
|
1580
1613
|
};
|
|
1581
1614
|
const stopMeetingFlush = () => {
|
|
1582
1615
|
if (meetingFlushTimer) {
|
|
@@ -1633,8 +1666,40 @@ async function main() {
|
|
|
1633
1666
|
// is empty until the meeting ENDS, so mid-call the LLM saw nothing.
|
|
1634
1667
|
recall.on('transcript', ({ botId, speaker, text, partial }) => {
|
|
1635
1668
|
console.log(`📝 Meeting transcript [${speaker}]${partial ? ' (partial)' : ''}: ${text}`);
|
|
1636
|
-
if (!partial && text.trim())
|
|
1669
|
+
if (!partial && text.trim()) {
|
|
1637
1670
|
meetingTranscriptBuffer.push(`${speaker}: ${text.trim()}`);
|
|
1671
|
+
// Addressed by name → flush NOW (don't wait the 20s batch) and tag it so
|
|
1672
|
+
// the agent replies out loud into the meeting. This is the chat-mode path:
|
|
1673
|
+
// named/asked → prompt response. Un-addressed chunks stay in the silent
|
|
1674
|
+
// 20s batch for note-taking. That's the two-mode seam, mechanically.
|
|
1675
|
+
if (/\bosborn\b/i.test(text)) {
|
|
1676
|
+
console.log('📓 Addressed by name — immediate flush for a response');
|
|
1677
|
+
flushMeetingBuffer(botId, true);
|
|
1678
|
+
}
|
|
1679
|
+
}
|
|
1680
|
+
});
|
|
1681
|
+
// Participant speech VAD → interruption + silence-boundary chunking.
|
|
1682
|
+
recall.on('speech', ({ botId, participant, active }) => {
|
|
1683
|
+
const isBot = participant === meetingBotName || participant === 'Osborn' || participant === 'Unknown';
|
|
1684
|
+
if (active) {
|
|
1685
|
+
// A HUMAN started talking. If the bot is mid-speech → INTERRUPT: stop the
|
|
1686
|
+
// bot's audio and record what it was cut off saying so the next flush can
|
|
1687
|
+
// tell it what it missed (same pattern as voice-native interruptions).
|
|
1688
|
+
if (meetingAgentSpeaking && !isBot) {
|
|
1689
|
+
console.log(`✋ Interruption — ${participant} spoke while bot was talking. Stopping bot audio.`);
|
|
1690
|
+
pushCanvas({ kind: 'stop' });
|
|
1691
|
+
meetingInterruptContext = `[MEETING — interrupted] You were speaking ("${meetingAgentSpeakingText.slice(0, 140)}") when ${participant} started talking and cut you off. They likely didn't hear the rest. When you respond, briefly acknowledge and adapt — don't just repeat.`;
|
|
1692
|
+
meetingAgentSpeaking = false;
|
|
1693
|
+
if (meetingSpeakClearTimer)
|
|
1694
|
+
clearTimeout(meetingSpeakClearTimer);
|
|
1695
|
+
}
|
|
1696
|
+
}
|
|
1697
|
+
else {
|
|
1698
|
+
// speech_off — a natural silence boundary. Flush the buffered turns now
|
|
1699
|
+
// (chunk on conversational pauses, not an arbitrary clock), unless empty.
|
|
1700
|
+
if (!isBot && meetingTranscriptBuffer.length)
|
|
1701
|
+
flushMeetingBuffer(botId, false);
|
|
1702
|
+
}
|
|
1638
1703
|
});
|
|
1639
1704
|
}
|
|
1640
1705
|
// ============================================================
|
|
@@ -2287,6 +2352,14 @@ async function main() {
|
|
|
2287
2352
|
console.log(`🔇 tts_say fired but text is empty — skipping`);
|
|
2288
2353
|
return;
|
|
2289
2354
|
}
|
|
2355
|
+
// Suppress browser TTS for meeting-chunk responses — the agent speaks INTO
|
|
2356
|
+
// the meeting via /canvas, not the laptop speakers (which the Meet mic
|
|
2357
|
+
// re-captures in the same room → feedback). Set by PipelineDirectLLM.chat()
|
|
2358
|
+
// when the turn is a [MEETING —] chunk. Normal user turns are unaffected.
|
|
2359
|
+
if (directLLM.suppressMeetingTTS) {
|
|
2360
|
+
console.log(`🔇 tts_say suppressed (meeting turn — response goes to /canvas, not browser): "${data.text.slice(0, 60)}"`);
|
|
2361
|
+
return;
|
|
2362
|
+
}
|
|
2290
2363
|
// Suppress while the user is mid-utterance. Without this, agent text generated
|
|
2291
2364
|
// in parallel by the Claude SDK plays right over the user — same problem as
|
|
2292
2365
|
// pre-interrupt overlap, but at the *output* side. The suppressed text gets
|
|
@@ -42,6 +42,7 @@ export interface FastBrainPanelResult {
|
|
|
42
42
|
}
|
|
43
43
|
export declare class PipelineDirectLLM extends llm.LLM {
|
|
44
44
|
#private;
|
|
45
|
+
suppressMeetingTTS: boolean;
|
|
45
46
|
constructor(opts: PipelineDirectOptions);
|
|
46
47
|
/** Stop the index watcher (call on disconnect/session switch) */
|
|
47
48
|
stopIndexWatcher(): void;
|
|
@@ -18,6 +18,10 @@ export class PipelineDirectLLM extends llm.LLM {
|
|
|
18
18
|
#turnAbort = null;
|
|
19
19
|
#indexWatcher = null;
|
|
20
20
|
#indexBuilding = false;
|
|
21
|
+
// True while the in-flight turn is a [MEETING —] chunk — the tts_say gate in
|
|
22
|
+
// index.ts reads this to skip browser session.say() (meeting audio goes via
|
|
23
|
+
// /canvas, not the laptop speakers → no same-room feedback). Set per chat().
|
|
24
|
+
suppressMeetingTTS = false;
|
|
21
25
|
constructor(opts) {
|
|
22
26
|
super();
|
|
23
27
|
this.#claudeLLM = new ClaudeLLM(opts);
|
|
@@ -175,10 +179,20 @@ export class PipelineDirectLLM extends llm.LLM {
|
|
|
175
179
|
}
|
|
176
180
|
}
|
|
177
181
|
}
|
|
182
|
+
// Meeting chunks ([MEETING —]) are the silent-observer / addressed-response
|
|
183
|
+
// path. They must NOT drive the browser voice session: (a) no session.say()
|
|
184
|
+
// TTS — the Meet mic re-captures browser audio in the same room → feedback;
|
|
185
|
+
// the agent speaks INTO the meeting via /canvas instead; (b) no fast brain —
|
|
186
|
+
// it's chat-panel noise ("Is there something you'd like to know?") on meeting
|
|
187
|
+
// speech. Set BEFORE the response streams so the tts_say gate (index.ts) sees
|
|
188
|
+
// it. Cleared implicitly by the next real user turn (which isn't a [MEETING]).
|
|
189
|
+
const isMeetingChunk = userText.startsWith('[MEETING');
|
|
190
|
+
this.suppressMeetingTTS = isMeetingChunk;
|
|
178
191
|
// Fire Claude
|
|
179
192
|
const claudeStream = this.#claudeLLM.chat({ chatCtx, toolCtx, connOptions, abortController });
|
|
180
|
-
// Fire pipeline fast brain in background — no await, no blocking
|
|
181
|
-
|
|
193
|
+
// Fire pipeline fast brain in background — no await, no blocking. Skip for
|
|
194
|
+
// meeting chunks (silent observer / meeting-response path, not a user turn).
|
|
195
|
+
if (userText.trim() && !isMeetingChunk) {
|
|
182
196
|
this.#firePipelineFastBrain(userText);
|
|
183
197
|
}
|
|
184
198
|
return claudeStream;
|
package/dist/recall-client.js
CHANGED
|
@@ -45,8 +45,12 @@ export class RecallClient extends EventEmitter {
|
|
|
45
45
|
// transcript turn to /webhook/recall as it's spoken — the receiver
|
|
46
46
|
// (handleWebhook) already exists. Polling stays as the after-the-fact
|
|
47
47
|
// backstop. Only wire the webhook when we have a public URL to receive on.
|
|
48
|
+
// Subscribe to transcript (content) AND participant speech VAD events. The
|
|
49
|
+
// speech_on/speech_off events fire from raw audio (faster than transcription)
|
|
50
|
+
// and tell us WHO is speaking WHEN — used for interruption (human speaks while
|
|
51
|
+
// the bot is talking → stop) and for chunking on natural silence boundaries.
|
|
48
52
|
const realtime = /^https:\/\//.test(webhookBaseUrl)
|
|
49
|
-
? [{ type: 'webhook', url: `${webhookBaseUrl}/webhook/recall`, events: ['transcript.data', 'transcript.partial_data'] }]
|
|
53
|
+
? [{ type: 'webhook', url: `${webhookBaseUrl}/webhook/recall`, events: ['transcript.data', 'transcript.partial_data', 'participant_events.speech_on', 'participant_events.speech_off'] }]
|
|
50
54
|
: [];
|
|
51
55
|
if (!realtime.length)
|
|
52
56
|
console.log('⚠️ Recall realtime webhook skipped (no public https URL) — polling only');
|
|
@@ -154,6 +158,21 @@ export class RecallClient extends EventEmitter {
|
|
|
154
158
|
// grows meaningfully or finalizes.
|
|
155
159
|
#lastEmitted = new Map();
|
|
156
160
|
handleWebhook(payload) {
|
|
161
|
+
// Participant speech VAD (speech_on/speech_off) — fires from raw audio, ahead
|
|
162
|
+
// of transcription. Emit a 'speech' event so index.ts can (a) interrupt the
|
|
163
|
+
// bot when a HUMAN starts talking over it, and (b) chunk the transcript on
|
|
164
|
+
// natural silence boundaries. The bot's own output_media audio may also fire
|
|
165
|
+
// speech_on for its own participant — index.ts filters that by name.
|
|
166
|
+
if (payload.event === 'participant_events.speech_on' || payload.event === 'participant_events.speech_off') {
|
|
167
|
+
const p = payload.data;
|
|
168
|
+
this.emit('speech', {
|
|
169
|
+
botId: p?.bot?.id ?? 'unknown',
|
|
170
|
+
participant: p?.data?.participant?.name ?? 'Unknown',
|
|
171
|
+
isHost: !!p?.data?.participant?.is_host,
|
|
172
|
+
active: payload.event === 'participant_events.speech_on',
|
|
173
|
+
});
|
|
174
|
+
return;
|
|
175
|
+
}
|
|
157
176
|
// Accept BOTH finals (transcript.data) AND partials (transcript.partial_data).
|
|
158
177
|
// recallai_streaming in prioritize_low_latency mode emits partials DURING
|
|
159
178
|
// the call and finals lag — ignoring partials (the old behavior) meant no
|