osborn 0.9.120 → 0.9.121
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +121 -26
- package/dist/recall-client.d.ts +8 -0
- package/dist/recall-client.js +25 -0
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -235,6 +235,38 @@ let meetingAgentSpeakingText = '';
|
|
|
235
235
|
// Prepended to the next flush: what the bot was cut off saying + who interrupted
|
|
236
236
|
// (same pattern as voice-native interruptions).
|
|
237
237
|
let meetingInterruptContext = '';
|
|
238
|
+
// Meeting speech QUEUE (0.9.121): serialize output_audio so replies never
|
|
239
|
+
// overlap — the Recall-sink equivalent of session.say's SpeechHandle queue.
|
|
240
|
+
// Recall's output_audio POST returns on ACCEPT, not on finish, so without this
|
|
241
|
+
// two replies (from two flushes, or a streamed multi-chunk reply) play ON TOP
|
|
242
|
+
// of each other — the "another voice over it" the user heard. Each utterance
|
|
243
|
+
// waits for the prior one's estimated playback before it plays; a generation
|
|
244
|
+
// counter (bumped on human interruption / a superseding turn) discards anything
|
|
245
|
+
// still queued so the bot never talks over itself or a human.
|
|
246
|
+
let meetingSpeakChain = Promise.resolve();
|
|
247
|
+
let meetingSpeakGen = 0;
|
|
248
|
+
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
249
|
+
// Estimated playback duration of a spoken line: ~2.5 words/sec + ~0.8s Recall
|
|
250
|
+
// buffer. Used to hold the speech queue so the next utterance doesn't overlap.
|
|
251
|
+
function estimatedSpeechMs(text) {
|
|
252
|
+
const words = text.split(/\s+/).filter(Boolean).length;
|
|
253
|
+
return Math.min(30_000, 800 + (words / 2.5) * 1000);
|
|
254
|
+
}
|
|
255
|
+
// Interrupt all meeting speech: bump the generation (drops queued + in-synth
|
|
256
|
+
// utterances) and stop any output_audio Recall is currently playing.
|
|
257
|
+
function interruptMeetingSpeech(reason) {
|
|
258
|
+
meetingSpeakGen++;
|
|
259
|
+
meetingAgentSpeaking = false;
|
|
260
|
+
if (meetingSpeakClearTimer) {
|
|
261
|
+
clearTimeout(meetingSpeakClearTimer);
|
|
262
|
+
meetingSpeakClearTimer = null;
|
|
263
|
+
}
|
|
264
|
+
const recall = getRecallClient();
|
|
265
|
+
const botId = recall?.getActiveBotIds?.()[0];
|
|
266
|
+
if (recall && botId)
|
|
267
|
+
void recall.stopOutputAudio(botId);
|
|
268
|
+
console.log(`✋ meeting speech interrupted (${reason}) — queue cleared + output_audio stopped`);
|
|
269
|
+
}
|
|
238
270
|
// Synthesize speech as MP3 (Deepgram fast path, OpenAI fallback) — for
|
|
239
271
|
// Recall native output_audio, which requires mp3.
|
|
240
272
|
async function synthMp3(text) {
|
|
@@ -282,21 +314,43 @@ async function synthMp3(text) {
|
|
|
282
314
|
// bot's camera shows what it's saying — no double-audio. output_audio +
|
|
283
315
|
// canvas camera coexist (confirmed: user heard output_audio while the canvas
|
|
284
316
|
// was showing). Falls back to canvas 'say' (audio) only if output_audio fails.
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
317
|
+
function speakIntoMeeting(text) {
|
|
318
|
+
if (!text?.trim())
|
|
319
|
+
return Promise.resolve();
|
|
320
|
+
// Capture the generation at ENQUEUE time. If an interrupt (or a superseding
|
|
321
|
+
// turn) bumps the gen before this item runs — or mid-synth — we drop it, so
|
|
322
|
+
// the bot never plays a reply the conversation has already moved past.
|
|
323
|
+
const gen = meetingSpeakGen;
|
|
324
|
+
const run = meetingSpeakChain.then(async () => {
|
|
325
|
+
if (gen !== meetingSpeakGen) {
|
|
326
|
+
console.log(`🔇 meeting speech superseded — dropping: "${text.slice(0, 40)}"`);
|
|
294
327
|
return;
|
|
295
328
|
}
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
329
|
+
const recall = getRecallClient();
|
|
330
|
+
const botId = recall?.getActiveBotIds?.()[0];
|
|
331
|
+
if (recall && botId) {
|
|
332
|
+
const mp3 = await synthMp3(text);
|
|
333
|
+
if (gen !== meetingSpeakGen) {
|
|
334
|
+
console.log(`🔇 meeting speech interrupted mid-synth — dropping: "${text.slice(0, 40)}"`);
|
|
335
|
+
return;
|
|
336
|
+
}
|
|
337
|
+
if (mp3 && await recall.outputAudio(botId, mp3)) {
|
|
338
|
+
console.log(`📢 spoke via Recall output_audio (${mp3.length}b): "${text.slice(0, 60)}"`);
|
|
339
|
+
pushCanvas({ kind: 'caption', text }); // visual only, no audio
|
|
340
|
+
markMeetingSpeaking(text);
|
|
341
|
+
// Hold the queue for the estimated playback so the NEXT utterance
|
|
342
|
+
// doesn't start on top of this one (POST returns on accept, not finish).
|
|
343
|
+
await sleep(estimatedSpeechMs(text));
|
|
344
|
+
return;
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
console.log(`📽️ falling back to canvas say (audio): "${text.slice(0, 50)}"`);
|
|
348
|
+
pushCanvas({ kind: 'say', text });
|
|
349
|
+
markMeetingSpeaking(text);
|
|
350
|
+
await sleep(estimatedSpeechMs(text));
|
|
351
|
+
}).catch((e) => { console.warn(`⚠️ meeting speak failed: ${e.message}`); });
|
|
352
|
+
meetingSpeakChain = run;
|
|
353
|
+
return run;
|
|
300
354
|
}
|
|
301
355
|
function markMeetingSpeaking(text) {
|
|
302
356
|
meetingAgentSpeaking = true;
|
|
@@ -1775,12 +1829,12 @@ async function main() {
|
|
|
1775
1829
|
// from muting the audio path. Reset by endMeeting.
|
|
1776
1830
|
meetingAddressedUntil = Date.now() + 6 * 60 * 60 * 1000;
|
|
1777
1831
|
}
|
|
1778
|
-
//
|
|
1779
|
-
//
|
|
1780
|
-
//
|
|
1781
|
-
//
|
|
1832
|
+
// Addressed turns get a MINIMAL tag + a hard brevity rule (0.9.121): the
|
|
1833
|
+
// reply is SPOKEN into a live meeting, so long answers (a) take 8s+ to
|
|
1834
|
+
// synthesize and (b) pile up / talk over the next turn. 1–2 sentences keeps
|
|
1835
|
+
// it conversational and fast; the agent can offer to go deeper if asked.
|
|
1782
1836
|
const header = addressed
|
|
1783
|
-
? `[MEETING — ${botId}] (addressed — reply is
|
|
1837
|
+
? `[MEETING — ${botId}] (addressed — your reply is SPOKEN OUT LOUD into the meeting: keep it to 1–2 short sentences, conversational, no lists/markdown; offer to elaborate only if they want more):`
|
|
1784
1838
|
: `[MEETING — ${botId}]:`;
|
|
1785
1839
|
// Prepend + consume any interruption context (bot was cut off mid-sentence).
|
|
1786
1840
|
const interrupt = meetingInterruptContext ? `${meetingInterruptContext}\n` : '';
|
|
@@ -1806,8 +1860,33 @@ async function main() {
|
|
|
1806
1860
|
meetingFlushTimer = null;
|
|
1807
1861
|
console.log('📓 Meeting flush timer stopped');
|
|
1808
1862
|
}
|
|
1863
|
+
if (addressedFlushTimer) {
|
|
1864
|
+
clearTimeout(addressedFlushTimer);
|
|
1865
|
+
addressedFlushTimer = null;
|
|
1866
|
+
}
|
|
1809
1867
|
meetingTranscriptBuffer.length = 0;
|
|
1810
1868
|
};
|
|
1869
|
+
// TURN-DEBOUNCED addressed flush (0.9.121): Recall closes a transcript
|
|
1870
|
+
// segment on every pause, so flushing on each final made the bot reply to
|
|
1871
|
+
// FRAGMENTS mid-thought (and multiple times per utterance). Instead we
|
|
1872
|
+
// debounce: each new transcript final resets a short timer; we only flush
|
|
1873
|
+
// (= reply) after the speaker has actually paused, so one turn = one reply.
|
|
1874
|
+
// speech_off (a hard silence boundary) can flush sooner via a smaller delay.
|
|
1875
|
+
let addressedFlushTimer = null;
|
|
1876
|
+
const ADDRESSED_DEBOUNCE_MS = 1400;
|
|
1877
|
+
const scheduleAddressedFlush = (botId, delayMs = ADDRESSED_DEBOUNCE_MS) => {
|
|
1878
|
+
if (addressedFlushTimer)
|
|
1879
|
+
clearTimeout(addressedFlushTimer);
|
|
1880
|
+
addressedFlushTimer = setTimeout(() => {
|
|
1881
|
+
addressedFlushTimer = null;
|
|
1882
|
+
if (!meetingTranscriptBuffer.length)
|
|
1883
|
+
return;
|
|
1884
|
+
// A fresh human turn supersedes anything the bot was still saying about
|
|
1885
|
+
// the previous one — interrupt stale queued/playing speech before we reply.
|
|
1886
|
+
interruptMeetingSpeech('new addressed turn');
|
|
1887
|
+
flushMeetingBuffer(botId, true);
|
|
1888
|
+
}, delayMs);
|
|
1889
|
+
};
|
|
1811
1890
|
// ── Meeting lifecycle (centralized teardown, 0.9.95) ──
|
|
1812
1891
|
// The bot's lifecycle FOLLOWS the voice session (deliberate coupling — a
|
|
1813
1892
|
// decoupled always-on bot means untracked background agents; revisit only
|
|
@@ -1969,8 +2048,12 @@ async function main() {
|
|
|
1969
2048
|
}
|
|
1970
2049
|
const oneOnOne = meetingSpeakers.size <= 1;
|
|
1971
2050
|
if (oneOnOne || /\b(osborne?|oz\s?born|os\s?born|was born|is born|ozborn|osbourne?|austin\b.{0,8}(hear|there|can you))/i.test(text)) {
|
|
1972
|
-
|
|
1973
|
-
|
|
2051
|
+
// DEBOUNCED (0.9.121): don't reply to this fragment — wait for the
|
|
2052
|
+
// speaker to actually pause. Each new final resets the timer, so one
|
|
2053
|
+
// continuous thought (even across Recall's mid-sentence segment splits)
|
|
2054
|
+
// becomes ONE reply instead of several talking over each other.
|
|
2055
|
+
console.log(`📓 Addressed (${oneOnOne ? '1:1 meeting' : 'by name'}) — turn debounced (~${ADDRESSED_DEBOUNCE_MS}ms)`);
|
|
2056
|
+
scheduleAddressedFlush(botId);
|
|
1974
2057
|
}
|
|
1975
2058
|
}
|
|
1976
2059
|
});
|
|
@@ -1982,12 +2065,16 @@ async function main() {
|
|
|
1982
2065
|
// bot's audio and record what it was cut off saying so the next flush can
|
|
1983
2066
|
// tell it what it missed (same pattern as voice-native interruptions).
|
|
1984
2067
|
if (meetingAgentSpeaking && !isBot) {
|
|
1985
|
-
console.log(`✋ Interruption — ${participant} spoke while bot was talking
|
|
1986
|
-
|
|
2068
|
+
console.log(`✋ Interruption — ${participant} spoke while bot was talking.`);
|
|
2069
|
+
// Capture what got cut off BEFORE interrupting (same interruption-
|
|
2070
|
+
// context ledger as the website path — the next flush tells the agent
|
|
2071
|
+
// it was cut off + what the human likely didn't hear).
|
|
1987
2072
|
meetingInterruptContext = `[MEETING — interrupted] You were speaking ("${meetingAgentSpeakingText.slice(0, 140)}") when ${participant} started talking and cut you off. They likely didn't hear the rest. When you respond, briefly acknowledge and adapt — don't just repeat.`;
|
|
1988
|
-
|
|
1989
|
-
|
|
1990
|
-
|
|
2073
|
+
// Actually STOP the voice: canvas stop + Recall output_audio stop +
|
|
2074
|
+
// queue-generation bump (drops anything still queued). Before 0.9.121
|
|
2075
|
+
// only the canvas was stopped — the real output_audio kept playing.
|
|
2076
|
+
pushCanvas({ kind: 'stop' });
|
|
2077
|
+
interruptMeetingSpeech(`human ${participant} barged in`);
|
|
1991
2078
|
}
|
|
1992
2079
|
}
|
|
1993
2080
|
else {
|
|
@@ -1998,7 +2085,15 @@ async function main() {
|
|
|
1998
2085
|
// (user directive 2026-08-01).
|
|
1999
2086
|
if (!isBot && meetingTranscriptBuffer.length) {
|
|
2000
2087
|
const latchOpen = Date.now() < meetingAddressedUntil || meetingSpeakers.size <= 1;
|
|
2001
|
-
|
|
2088
|
+
if (latchOpen) {
|
|
2089
|
+
// Hard silence boundary → the speaker really finished. Flush sooner
|
|
2090
|
+
// than the transcript debounce (still debounced so back-to-back
|
|
2091
|
+
// speakers coalesce into one turn).
|
|
2092
|
+
scheduleAddressedFlush(botId, 450);
|
|
2093
|
+
}
|
|
2094
|
+
else {
|
|
2095
|
+
flushMeetingBuffer(botId, false); // silent observer note-taking batch
|
|
2096
|
+
}
|
|
2002
2097
|
}
|
|
2003
2098
|
}
|
|
2004
2099
|
});
|
package/dist/recall-client.d.ts
CHANGED
|
@@ -110,6 +110,14 @@ export declare class RecallClient extends EventEmitter {
|
|
|
110
110
|
* was "barely bearable"; this is the direct loud path.)
|
|
111
111
|
*/
|
|
112
112
|
outputAudio(botId: string, mp3: Buffer): Promise<boolean>;
|
|
113
|
+
/**
|
|
114
|
+
* Stop the bot's currently-playing output_audio (barge-in / interruption).
|
|
115
|
+
* Recall exposes DELETE on the same endpoint to clear in-progress playback.
|
|
116
|
+
* Best-effort: returns true on 2xx, false otherwise (older bots / no active
|
|
117
|
+
* audio may 404 — harmless, the speech queue's generation bump still halts
|
|
118
|
+
* anything queued). Mirrors the website path's TTS interrupt on barge-in.
|
|
119
|
+
*/
|
|
120
|
+
stopOutputAudio(botId: string): Promise<boolean>;
|
|
113
121
|
getBotStatus(botId: string): Promise<string>;
|
|
114
122
|
handleWebhook(payload: TranscriptPayload): void;
|
|
115
123
|
registerBot(botId: string, sessionId: string): void;
|
package/dist/recall-client.js
CHANGED
|
@@ -165,6 +165,31 @@ export class RecallClient extends EventEmitter {
|
|
|
165
165
|
}
|
|
166
166
|
return true;
|
|
167
167
|
}
|
|
168
|
+
/**
|
|
169
|
+
* Stop the bot's currently-playing output_audio (barge-in / interruption).
|
|
170
|
+
* Recall exposes DELETE on the same endpoint to clear in-progress playback.
|
|
171
|
+
* Best-effort: returns true on 2xx, false otherwise (older bots / no active
|
|
172
|
+
* audio may 404 — harmless, the speech queue's generation bump still halts
|
|
173
|
+
* anything queued). Mirrors the website path's TTS interrupt on barge-in.
|
|
174
|
+
*/
|
|
175
|
+
async stopOutputAudio(botId) {
|
|
176
|
+
try {
|
|
177
|
+
const res = await fetch(`${RECALL_BASE_URL}/bot/${botId}/output_audio/`, {
|
|
178
|
+
method: 'DELETE',
|
|
179
|
+
headers: { 'Authorization': `Token ${this.#apiKey}` },
|
|
180
|
+
});
|
|
181
|
+
if (!res.ok) {
|
|
182
|
+
const e = await res.text().catch(() => '');
|
|
183
|
+
console.warn(`⚠️ Recall stop output_audio ${res.status}: ${e.slice(0, 120)}`);
|
|
184
|
+
return false;
|
|
185
|
+
}
|
|
186
|
+
return true;
|
|
187
|
+
}
|
|
188
|
+
catch (err) {
|
|
189
|
+
console.warn(`⚠️ Recall stop output_audio failed: ${err.message}`);
|
|
190
|
+
return false;
|
|
191
|
+
}
|
|
192
|
+
}
|
|
168
193
|
async getBotStatus(botId) {
|
|
169
194
|
const res = await fetch(`${RECALL_BASE_URL}/bot/${botId}`, {
|
|
170
195
|
headers: { 'Authorization': `Token ${this.#apiKey}` },
|