osborn 0.9.106 → 0.9.108
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +51 -5
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -570,11 +570,45 @@ function startApiServer(workingDir, port) {
|
|
|
570
570
|
// speechSynthesis is NOT captured by Recall, a media element IS.
|
|
571
571
|
if (req.method === 'GET' && url.pathname === '/tts') {
|
|
572
572
|
const text = (url.searchParams.get('text') || '').slice(0, 4000);
|
|
573
|
+
if (!text) {
|
|
574
|
+
res.writeHead(400, { 'Content-Type': 'application/json' });
|
|
575
|
+
res.end(JSON.stringify({ error: 'no text' }));
|
|
576
|
+
return;
|
|
577
|
+
}
|
|
578
|
+
const t0 = Date.now();
|
|
579
|
+
// Deepgram FIRST (2026-08-01 latency fix): same TTS family the regular
|
|
580
|
+
// voice pipeline uses — ~3-6x faster than the OpenAI full-file synth this
|
|
581
|
+
// endpoint used before (measured 2-4s of the meeting reply lag). OpenAI
|
|
582
|
+
// stays as fallback.
|
|
583
|
+
const dgKey = process.env.DEEPGRAM_API_KEY;
|
|
584
|
+
if (dgKey) {
|
|
585
|
+
try {
|
|
586
|
+
// WAV/linear16, not mp3: mp3 files carry encoder padding (leading/
|
|
587
|
+
// trailing silence + boundary click) — back-to-back sentence clips
|
|
588
|
+
// were heard as "cracking". WAV is gapless-safe and cheaper to decode.
|
|
589
|
+
const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-thalia-en&encoding=linear16&sample_rate=24000&container=wav', {
|
|
590
|
+
method: 'POST',
|
|
591
|
+
headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
|
|
592
|
+
body: JSON.stringify({ text }),
|
|
593
|
+
});
|
|
594
|
+
if (dg.ok) {
|
|
595
|
+
const buf = Buffer.from(await dg.arrayBuffer());
|
|
596
|
+
console.log(`🗣️ /tts deepgram wav ${buf.length}b in ${Date.now() - t0}ms (${text.length} chars) t=${new Date().toISOString()}`);
|
|
597
|
+
res.writeHead(200, { 'Content-Type': 'audio/wav', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
|
|
598
|
+
res.end(buf);
|
|
599
|
+
return;
|
|
600
|
+
}
|
|
601
|
+
console.warn(`⚠️ /tts deepgram ${dg.status} — falling back to OpenAI`);
|
|
602
|
+
}
|
|
603
|
+
catch (e) {
|
|
604
|
+
console.warn(`⚠️ /tts deepgram error: ${e.message} — falling back to OpenAI`);
|
|
605
|
+
}
|
|
606
|
+
}
|
|
573
607
|
const voice = url.searchParams.get('voice') || 'alloy';
|
|
574
608
|
const key = process.env.OPENAI_API_KEY;
|
|
575
|
-
if (!
|
|
609
|
+
if (!key) {
|
|
576
610
|
res.writeHead(400, { 'Content-Type': 'application/json' });
|
|
577
|
-
res.end(JSON.stringify({ error:
|
|
611
|
+
res.end(JSON.stringify({ error: 'no TTS provider keys' }));
|
|
578
612
|
return;
|
|
579
613
|
}
|
|
580
614
|
try {
|
|
@@ -590,6 +624,7 @@ function startApiServer(workingDir, port) {
|
|
|
590
624
|
return;
|
|
591
625
|
}
|
|
592
626
|
const buf = Buffer.from(await tts.arrayBuffer());
|
|
627
|
+
console.log(`🗣️ /tts openai ${buf.length}b in ${Date.now() - t0}ms t=${new Date().toISOString()}`);
|
|
593
628
|
res.writeHead(200, { 'Content-Type': 'audio/mpeg', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
|
|
594
629
|
res.end(buf);
|
|
595
630
|
}
|
|
@@ -1641,8 +1676,12 @@ async function main() {
|
|
|
1641
1676
|
// reply entirely) — measured as most of the 5-8s reply lag.
|
|
1642
1677
|
meetingAddressedUntil = Date.now() + 90_000;
|
|
1643
1678
|
}
|
|
1679
|
+
// PROMPT PARITY (user directive 2026-08-01): addressed turns carry ONLY a
|
|
1680
|
+
// minimal tag — no behavioral re-instruction. The agent replies exactly as
|
|
1681
|
+
// it would to a regular voice turn; the tts_say→canvas redirect handles
|
|
1682
|
+
// where the words go. Same agent, same behavior, both fronts.
|
|
1644
1683
|
const header = addressed
|
|
1645
|
-
? `[MEETING — ${botId}]
|
|
1684
|
+
? `[MEETING — ${botId}] (addressed — reply is spoken into the meeting):`
|
|
1646
1685
|
: `[MEETING — ${botId}]:`;
|
|
1647
1686
|
// Prepend + consume any interruption context (bot was cut off mid-sentence).
|
|
1648
1687
|
const interrupt = meetingInterruptContext ? `${meetingInterruptContext}\n` : '';
|
|
@@ -2511,8 +2550,15 @@ async function main() {
|
|
|
2511
2550
|
// (reuse of the regular speak path; no Bash roundtrip). Silent-observer
|
|
2512
2551
|
// turns stay suppressed as before.
|
|
2513
2552
|
if (activeMeetingBotId && Date.now() < meetingAddressedUntil) {
|
|
2514
|
-
|
|
2515
|
-
|
|
2553
|
+
// Split into AT MOST 2 chunks: first sentence (fast first-audio) +
|
|
2554
|
+
// the remainder as ONE chunk. Per-sentence files created an audible
|
|
2555
|
+
// seam at every boundary (mp3 padding + scheduling jitter — heard as
|
|
2556
|
+
// "cracking"); two chunks keeps the latency win with one seam max.
|
|
2557
|
+
const m = data.text.match(/^([^.!?]+[.!?]+["']?)\s*([\s\S]*)$/);
|
|
2558
|
+
const chunks = m && m[2]?.trim() ? [m[1].trim(), m[2].trim()] : [data.text.trim()];
|
|
2559
|
+
console.log(`🔊➡️📽️ tts_say → canvas (${chunks.length} chunk(s), addressed turn) t=${new Date().toISOString()}: "${data.text.slice(0, 60)}"`);
|
|
2560
|
+
for (const s of chunks)
|
|
2561
|
+
pushCanvas({ kind: 'say', text: s });
|
|
2516
2562
|
markMeetingSpeaking(data.text);
|
|
2517
2563
|
return;
|
|
2518
2564
|
}
|