osborn 0.9.105 → 0.9.107
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/skills/meetings/SKILL.md +11 -8
- package/dist/index.js +63 -3
- package/package.json +1 -1
|
@@ -91,16 +91,19 @@ When a chunk says **`YOU WERE ADDRESSED`**, people in the meeting are WAITING
|
|
|
91
91
|
for your voice. Measured failure (2026-08-01): notes-writing inside the reply
|
|
92
92
|
turn pushed speech→reply past 30 seconds ("very very delayed").
|
|
93
93
|
|
|
94
|
-
**Mandatory turn shape when addressed:**
|
|
95
|
-
1. **
|
|
96
|
-
|
|
97
|
-
|
|
94
|
+
**Mandatory turn shape when addressed (0.9.106+):**
|
|
95
|
+
1. **Just ANSWER in plain text — your words are spoken into the meeting
|
|
96
|
+
automatically.** The regular voice pipeline is redirected to the meeting
|
|
97
|
+
for addressed turns; do NOT use Bash/curl to speak (that adds a slow tool
|
|
98
|
+
roundtrip and can eat your tool budget). 1–2 short conversational
|
|
99
|
+
sentences, text first, before any tool call.
|
|
98
100
|
2. **THEN** delegate note-taking / research to the writer/researcher
|
|
99
101
|
sub-agents in the background as usual.
|
|
100
|
-
3. If you genuinely need a fact before answering,
|
|
101
|
-
("Good question — one second while I check"),
|
|
102
|
-
|
|
103
|
-
|
|
102
|
+
3. If you genuinely need a fact before answering, SAY a holding line first
|
|
103
|
+
("Good question — one second while I check"), look it up, then continue in
|
|
104
|
+
plain text. Never leave the room in silence while you work.
|
|
105
|
+
(The `/canvas say` curl remains for NON-addressed cases: when the voice-native
|
|
106
|
+
user says "tell the meeting X", or proactive announcements.)
|
|
104
107
|
|
|
105
108
|
## Browse requests while CASTING a live stream — drive the ENGINE, not the canvas
|
|
106
109
|
|
package/dist/index.js
CHANGED
|
@@ -570,11 +570,42 @@ function startApiServer(workingDir, port) {
|
|
|
570
570
|
// speechSynthesis is NOT captured by Recall, a media element IS.
|
|
571
571
|
if (req.method === 'GET' && url.pathname === '/tts') {
|
|
572
572
|
const text = (url.searchParams.get('text') || '').slice(0, 4000);
|
|
573
|
+
if (!text) {
|
|
574
|
+
res.writeHead(400, { 'Content-Type': 'application/json' });
|
|
575
|
+
res.end(JSON.stringify({ error: 'no text' }));
|
|
576
|
+
return;
|
|
577
|
+
}
|
|
578
|
+
const t0 = Date.now();
|
|
579
|
+
// Deepgram FIRST (2026-08-01 latency fix): same TTS family the regular
|
|
580
|
+
// voice pipeline uses — ~3-6x faster than the OpenAI full-file synth this
|
|
581
|
+
// endpoint used before (measured 2-4s of the meeting reply lag). OpenAI
|
|
582
|
+
// stays as fallback.
|
|
583
|
+
const dgKey = process.env.DEEPGRAM_API_KEY;
|
|
584
|
+
if (dgKey) {
|
|
585
|
+
try {
|
|
586
|
+
const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-thalia-en&encoding=mp3', {
|
|
587
|
+
method: 'POST',
|
|
588
|
+
headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
|
|
589
|
+
body: JSON.stringify({ text }),
|
|
590
|
+
});
|
|
591
|
+
if (dg.ok) {
|
|
592
|
+
const buf = Buffer.from(await dg.arrayBuffer());
|
|
593
|
+
console.log(`🗣️ /tts deepgram ${buf.length}b in ${Date.now() - t0}ms (${text.length} chars) t=${new Date().toISOString()}`);
|
|
594
|
+
res.writeHead(200, { 'Content-Type': 'audio/mpeg', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
|
|
595
|
+
res.end(buf);
|
|
596
|
+
return;
|
|
597
|
+
}
|
|
598
|
+
console.warn(`⚠️ /tts deepgram ${dg.status} — falling back to OpenAI`);
|
|
599
|
+
}
|
|
600
|
+
catch (e) {
|
|
601
|
+
console.warn(`⚠️ /tts deepgram error: ${e.message} — falling back to OpenAI`);
|
|
602
|
+
}
|
|
603
|
+
}
|
|
573
604
|
const voice = url.searchParams.get('voice') || 'alloy';
|
|
574
605
|
const key = process.env.OPENAI_API_KEY;
|
|
575
|
-
if (!
|
|
606
|
+
if (!key) {
|
|
576
607
|
res.writeHead(400, { 'Content-Type': 'application/json' });
|
|
577
|
-
res.end(JSON.stringify({ error:
|
|
608
|
+
res.end(JSON.stringify({ error: 'no TTS provider keys' }));
|
|
578
609
|
return;
|
|
579
610
|
}
|
|
580
611
|
try {
|
|
@@ -590,6 +621,7 @@ function startApiServer(workingDir, port) {
|
|
|
590
621
|
return;
|
|
591
622
|
}
|
|
592
623
|
const buf = Buffer.from(await tts.arrayBuffer());
|
|
624
|
+
console.log(`🗣️ /tts openai ${buf.length}b in ${Date.now() - t0}ms t=${new Date().toISOString()}`);
|
|
593
625
|
res.writeHead(200, { 'Content-Type': 'audio/mpeg', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
|
|
594
626
|
res.end(buf);
|
|
595
627
|
}
|
|
@@ -1617,6 +1649,10 @@ async function main() {
|
|
|
1617
1649
|
// gone forever, matching the skill-catalog pattern.
|
|
1618
1650
|
let userRemovedAgents = [];
|
|
1619
1651
|
let activeMeetingBotId = null; // Recall.ai bot ID if in a meeting
|
|
1652
|
+
// While set (Date.now() < value), the agent's streaming tts_say output is
|
|
1653
|
+
// REDIRECTED into the meeting canvas instead of suppressed — the regular
|
|
1654
|
+
// voice pipeline reused with the meeting as the sink (latency fix).
|
|
1655
|
+
let meetingAddressedUntil = 0;
|
|
1620
1656
|
let activeMeetingPoller = null; // Transcript poller bound to that bot
|
|
1621
1657
|
// LIVE meeting transcript → LLM (buffered webhook finals). See recall.on('transcript').
|
|
1622
1658
|
const meetingTranscriptBuffer = [];
|
|
@@ -1628,8 +1664,17 @@ async function main() {
|
|
|
1628
1664
|
if (!meetingTranscriptBuffer.length || !currentLLM)
|
|
1629
1665
|
return;
|
|
1630
1666
|
const turns = meetingTranscriptBuffer.splice(0); // drain
|
|
1667
|
+
if (addressed) {
|
|
1668
|
+
// REUSE the regular voice pipeline (2026-08-01 latency fix): during an
|
|
1669
|
+
// addressed turn the agent's normal streaming tts_say output is REDIRECTED
|
|
1670
|
+
// to the canvas (see the tts_say handler) — same speak path as regular
|
|
1671
|
+
// mode, sink = meeting. The old prompt made the agent compose a Bash curl
|
|
1672
|
+
// (extra LLM tool roundtrip, ate the 3-call budget, sometimes lost the
|
|
1673
|
+
// reply entirely) — measured as most of the 5-8s reply lag.
|
|
1674
|
+
meetingAddressedUntil = Date.now() + 90_000;
|
|
1675
|
+
}
|
|
1631
1676
|
const header = addressed
|
|
1632
|
-
? `[MEETING — ${botId}] — YOU WERE ADDRESSED.
|
|
1677
|
+
? `[MEETING — ${botId}] — YOU WERE ADDRESSED. Reply now in PLAIN TEXT — your words are spoken into the meeting automatically (do NOT use Bash/curl to speak). One or two short conversational sentences first; then, if needed, delegate notes/research in the background.`
|
|
1633
1678
|
: `[MEETING — ${botId}]:`;
|
|
1634
1679
|
// Prepend + consume any interruption context (bot was cut off mid-sentence).
|
|
1635
1680
|
const interrupt = meetingInterruptContext ? `${meetingInterruptContext}\n` : '';
|
|
@@ -2494,6 +2539,21 @@ async function main() {
|
|
|
2494
2539
|
// re-captures in the same room → feedback). Set by PipelineDirectLLM.chat()
|
|
2495
2540
|
// when the turn is a [MEETING —] chunk. Normal user turns are unaffected.
|
|
2496
2541
|
if (directLLM.suppressMeetingTTS) {
|
|
2542
|
+
// ADDRESSED turn → REDIRECT the normal streaming reply into the meeting
|
|
2543
|
+
// (reuse of the regular speak path; no Bash roundtrip). Silent-observer
|
|
2544
|
+
// turns stay suppressed as before.
|
|
2545
|
+
if (activeMeetingBotId && Date.now() < meetingAddressedUntil) {
|
|
2546
|
+
// Sentence-split so the FIRST sentence synthesizes + plays while the
|
|
2547
|
+
// rest queue behind it (canvas plays says sequentially + prefetches)
|
|
2548
|
+
// — first-audio latency = one short sentence's synth, not the whole
|
|
2549
|
+
// reply's.
|
|
2550
|
+
const sentences = data.text.match(/[^.!?]+[.!?]+["']?|[^.!?]+$/g)?.map((s) => s.trim()).filter(Boolean) || [data.text];
|
|
2551
|
+
console.log(`🔊➡️📽️ tts_say → canvas (${sentences.length} sentence(s), addressed turn) t=${new Date().toISOString()}: "${data.text.slice(0, 60)}"`);
|
|
2552
|
+
for (const s of sentences)
|
|
2553
|
+
pushCanvas({ kind: 'say', text: s });
|
|
2554
|
+
markMeetingSpeaking(data.text);
|
|
2555
|
+
return;
|
|
2556
|
+
}
|
|
2497
2557
|
console.log(`🔇 tts_say suppressed (meeting turn — response goes to /canvas, not browser): "${data.text.slice(0, 60)}"`);
|
|
2498
2558
|
return;
|
|
2499
2559
|
}
|